<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Genet.</journal-id>
<journal-title>Frontiers in Genetics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Genet.</abbrev-journal-title>
<issn pub-type="epub">1664-8021</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">859462</article-id>
<article-id pub-id-type="doi">10.3389/fgene.2022.859462</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Genetics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A Sparse Mixture-of-Experts Model With Screening of Genetic Associations to Guide Disease Subtyping</article-title>
<alt-title alt-title-type="left-running-head">Courbariaux et al.</alt-title>
<alt-title alt-title-type="right-running-head">Screening Genetic Associations for Subtyping</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Courbariaux</surname>
<given-names>Marie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1584678/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>De Santiago</surname>
<given-names>Kylliann</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1646094/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Dalmasso</surname>
<given-names>Cyril</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1182534/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Danjou</surname>
<given-names>Fabrice</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/233419/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bekadar</surname>
<given-names>Samir</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/942706/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Corvol</surname>
<given-names>Jean-Christophe</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Martinez</surname>
<given-names>Maria</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Szafranski</surname>
<given-names>Marie</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1646040/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ambroise</surname>
<given-names>Christophe</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/793514/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Universit&#xe9; Paris-Saclay</institution>, <institution>CNRS</institution>, <institution>Universit&#xe9; d&#x2019;&#xc9;vry</institution>, <institution>Laboratoire de Math&#xe9;matiques et Mod&#xe9;lisation d&#x2019;&#xc9;vry</institution>, <addr-line>&#xc9;vry-Courcouronnes</addr-line>, <country>France</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Sorbonne Universit&#xe9;</institution>, <institution>Paris Brain Institute&#x2013;ICM</institution>, <institution>Inserm</institution>, <institution>CNRS</institution>, <institution>Assistance Publique H&#xf4;pitaux de Paris</institution>, <institution>Piti&#xe9;-Salp&#xea;tri&#xe8;re Hospital</institution>, <institution>Department of Neurology</institution>, <addr-line>Paris</addr-line>, <country>France</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Institut de Recherche en Sant&#xe9; Digestive</institution>, <institution>Inserm</institution>, <institution>CHU Purpan</institution>, <addr-line>Toulouse</addr-line>, <country>France</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>ENSIIE</institution>, <addr-line>&#xc9;vry-Courcouronnes</addr-line>, <country>France</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/36872/overview">Min Zhang</ext-link>, Purdue University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/143825/overview">Shaoyu Li</ext-link>, University of North Carolina at Charlotte, United States</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/230426/overview">Doug Speed</ext-link>, Aarhus University, Denmark</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Marie Courbariaux, <email>marie.courbariaux@gmail.com</email>; Marie Szafranski, <email>marie.szafranski@math.cnrs.fr</email>
</corresp>
<fn fn-type="equal" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>These authors share last authorship</p>
</fn>
<fn fn-type="other">
<p>This article was submitted to Statistical Genetics and Methodology, a section of the journal Frontiers in Genetics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>06</day>
<month>06</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>859462</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>01</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>04</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Courbariaux, De Santiago, Dalmasso, Danjou, Bekadar, Corvol, Martinez, Szafranski and Ambroise.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Courbariaux, De Santiago, Dalmasso, Danjou, Bekadar, Corvol, Martinez, Szafranski and Ambroise</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>
<bold>Motivation:</bold> Identifying new genetic associations in non-Mendelian complex diseases is an increasingly difficult challenge. These diseases sometimes appear to have a significant component of heritability requiring explanation, and this missing heritability may be due to the existence of subtypes involving different genetic factors. Taking genetic information into account in clinical trials might potentially have a role in guiding the process of subtyping a complex disease. Most methods dealing with multiple sources of information rely on data transformation, and in disease subtyping, the two main strategies used are 1) the clustering of clinical data followed by posterior genetic analysis and 2) the concomitant clustering of clinical and genetic variables. Both of these strategies have limitations that we propose to address.</p>
<p>
<bold>Contribution:</bold> This work proposes an original method for disease subtyping on the basis of both longitudinal clinical variables and high-dimensional genetic markers <italic>via</italic> a sparse mixture-of-regressions model. The added value of our approach lies in its interpretability in relation to two aspects. First, our model links both clinical and genetic data with regard to their initial nature (i.e., without transformation) and does not require post-processing where the original information is accessed a second time to interpret the subtypes. Second, it can address large-scale problems because of a variable selection step that is used to discard genetic variables that may not be relevant for subtyping.</p>
<p>
<bold>Results:</bold> The proposed method was validated on simulations. A dataset from a cohort of Parkinson&#x2019;s disease patients was also analyzed. Several subtypes of the disease and genetic variants that potentially have a role in this typology were identified.</p>
<p>
<bold>Software availability:</bold> The <monospace>R</monospace> code for the proposed method, named <monospace>DiSuGen</monospace>, and a tutorial are available for download (see the references).</p>
</abstract>
<kwd-group>
<kwd>mixture of experts model</kwd>
<kwd>disease subtyping</kwd>
<kwd>clinical data</kwd>
<kwd>longitudinal data</kwd>
<kwd>genotyping</kwd>
<kwd>high dimension</kwd>
<kwd>variable selection</kwd>
<kwd>Parkinson&#x2019;s disease</kwd>
</kwd-group>
<contract-sponsor id="cn001">Agence Nationale de la Recherche<named-content content-type="fundref-id">10.13039/501100001665</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Known genetic markers in complex diseases usually account for only a part of calculated heritability. One possible explanation is that these complex diseases have different subtypes with different genetic factors. The identification of subtypes can nowadays draw upon large heterogeneous datasets, including patient follow-up and genotyping data.</p>
<p>When clinical and genomic information is available, subtyping can adopt either of two approaches: 1) the clustering of clinical data with a posterior genetic analysis, or 2) the concomitant clustering of clinical and genomic data. We will discuss the pros and cons of these two approaches in <xref ref-type="sec" rid="s2">Section 2</xref>.</p>
<sec id="s1-1">
<title>1.1 Contributions</title>
<p>In this work, we sketch a third way at the crossroads between the two approaches mentioned previously. This alternative approach consists in clustering the clinical variables by estimating a multinomial logistic regression model whose weights depend on the genetic variables. The model reflects the longitudinal nature of the clinical data and addresses the high dimensionality of the problem <italic>via</italic> a sparse constraint on the parameters involved in the logistic weights.</p>
</sec>
<sec id="s1-2">
<title>1.2 Organization of the Article</title>
<p>
<xref ref-type="sec" rid="s2">Section 2</xref> gives an overview of different strategies that may be used for disease subtyping where there are different sources of information. <xref ref-type="sec" rid="s3">Section 3</xref> proposes a framework related to mixture-of-experts models, for clustering clinical longitudinal data guided by genetic markers. <xref ref-type="sec" rid="s4">Section 4</xref> describes our proposed algorithm and its implementation in a high-dimensionality setting. <xref ref-type="sec" rid="s5">Section 5</xref> provides an illustration of our approach using numerical simulations, and <xref ref-type="sec" rid="s6">Section 6</xref> gives an analysis of a cohort of patients with Parkinson&#x2019;s disease.</p>
</sec>
</sec>
<sec id="s2">
<title>2 Disease Subtyping With Multiple Sources of Information</title>
<p>In this section, we briefly describe the various approaches used for clustering where there are different sources of data, focusing in particular on methods for disease subtyping with multiple information sources.</p>
<sec id="s2-1">
<title>2.1 Clustering of Clinical Data With Posterior Genetic Analysis</title>
<p>As outlined in the following, this is a two-step approach involving 1) disease subtyping based on clinical data, followed by 2) an analysis of the genetic associations in each subtype.</p>
<sec id="s2-1-1">
<title>2.1.1 Clustering of Clinical Data</title>
<p>The data often come from clinical follow-ups, and as such are generally longitudinal in nature. A review of clustering methods suitable for functional data, including longitudinal data, is discussed by <xref ref-type="bibr" rid="B23">Jacques and Preda (2014</xref>), with the following categorization:<list list-type="simple">
<list-item>
<p>&#x2022; <italic>Methods with a filtering step</italic> consist in characterizing the curves in terms of a few descriptors such as their slope and intercept, and then clustering on those descriptors.</p>
</list-item>
<list-item>
<p>&#x2022; <italic>Non-parametric methods</italic>, such as <italic>K</italic>-means, with distance metrics adapted to longitudinal data.</p>
</list-item>
<list-item>
<p>&#x2022; Finally, <italic>model-based methods</italic> appear to be the most suitable methods for the kind of short longitudinal data with numerous missing values that often arise from medical follow-ups. An overview of the approaches and tools devoted to <italic>mixture models</italic> for longitudinal data has been proposed by <xref ref-type="bibr" rid="B46">van der Nest et al. (2020</xref>).</p>
</list-item>
</list>
</p>
<p>
<italic>Remark.</italic> In this work, we focus on <italic>mixtures of experts</italic>, a specific category of <italic>mixture models</italic> (<xref ref-type="sec" rid="s3-1">Section 3.1</xref>).</p>
</sec>
<sec id="s2-1-2">
<title>2.1.2 Analysis of Clinical Clusters With Genomics</title>
<p>Following the clustering, this second step seeks to exhibit genetic associations underlying the clusters. One way of doing this would be to use the clusters as phenotypes in standard GWAS approaches that usually involve statistical procedures based on (multiple) hypothesis testing (<xref ref-type="bibr" rid="B2">Bush and Moore, 2012</xref>; <xref ref-type="bibr" rid="B20">Hayes, 2013</xref>). Another way would be to resort to classical supervised methods, such as (multinomial) logistic regression, with a feature selection procedure (<xref ref-type="bibr" rid="B29">Ma and Huang, 2008</xref>).</p>
</sec>
<sec id="s2-1-3">
<title>2.1.3 Limitation</title>
<p>Since the genetic analysis takes place only after the clustering of clinical data has been completed, the clustering step makes no reference to the genomic data. As a consequence, there can be no certainty regarding an association between the genomic information and the clinical clusters. Also, most sparse model-based clustering methods for high-dimensional functional or longitudinal data need to resort to dimensionality reduction techniques such as PCA or SVD, which are effective but present barriers to interpretation.</p>
</sec>
</sec>
<sec id="s2-2">
<title>2.2 Concomitant Clustering of Clinical and Genomic Data</title>
<p>Concomitant clustering using both clinical and genomic data represents an attractive alternative to the two-step approach described previously. However, a large number of variables may be present, meaning that feature- or variable-selection strategies are required to solve the problem.</p>
<sec id="s2-2-1">
<title>2.2.1 Multi-View Clustering</title>
<p>This framework, developed within the machine learning community, provides a number of popular methods for solving problems with different feature sets. The survey by <xref ref-type="bibr" rid="B11">Fu et al. (2020</xref>) groups these methods into three categories.<list list-type="simple">
<list-item>
<p>&#x2022; <italic>Graph-based methods</italic> combine different views according to their respective importance and then generally resort to spectral clustering algorithms.</p>
</list-item>
<list-item>
<p>&#x2022; <italic>Space-learning-based methods</italic> are designed to construct a new learning space using the most representative characteristic of each view to enhance clustering.</p>
</list-item>
<list-item>
<p>&#x2022; <italic>Binary-code-learning-based methods</italic> encode original data as binary features using mapping and reduction techniques to reduce computation time and memory use.</p>
</list-item>
</list>
</p>
<p>We also need to mention the <italic>Multiple Kernel Learning</italic> framework for clustering (<xref ref-type="bibr" rid="B48">Zhao et al., 2009</xref>), which corresponds to another kind of multi-view learning. In particular, <xref ref-type="bibr" rid="B32">Mariette and Villa-Vialaneix (2017</xref>) proposed (consensus) meta-kernels for aggregating different sources of information while preserving the original topology of the data. Among the various works devoted to disease subtyping using clinical and genomic information, those that come within the scope of multi-view clustering use <italic>space-learning-based methods</italic> with dimensionality reduction approaches. <xref ref-type="bibr" rid="B45">Sun et al. (2014</xref>) propose a multi-view co-clustering method based on Sparse Singular Value Decomposition (<xref ref-type="bibr" rid="B26">Lee et al., 2010</xref>). <xref ref-type="bibr" rid="B44">Sun et al. (2015</xref>) build on this work, providing convergence guarantees using the proximal alternating linearized minimization algorithm proposed by <xref ref-type="bibr" rid="B1">Bolte et al. (2014</xref>).</p>
</sec>
<sec id="s2-2-2">
<title>2.2.2 Integrative Clustering</title>
<p>In cancer research, a variety of statistical methodologies have emerged for analyzing data coming from different sources, generally multiple omics data, within the field of <italic>integrative genomics</italic> (<xref ref-type="bibr" rid="B25">Kristensen et al., 2014</xref>). The philosophy underlying these methodologies is closely related to multi-view learning. <xref ref-type="bibr" rid="B21">Huang et al. (2017</xref>) present a review of multi-omics integration tools. We must also mention <italic>mixOmics</italic> (<xref ref-type="bibr" rid="B39">Rohart et al., 2017</xref>), which proposes various sparse multivariate methods for exploring multiple omics datasets. More specifically, integrative clustering may be built on model-based approaches such as in the representative work by <xref ref-type="bibr" rid="B42">Shen et al. (2009)</xref> and <xref ref-type="bibr" rid="B43">Shen et al. (2010)</xref>. The <italic>iCluster</italic> method uses a latent variable model to connect multiple data types. The optimization of a penalized log-likelihood involves a process of dimensionality reduction on the representation of the original data that iteratively alternates with several extensions of <italic>iCluster</italic> using penalties inducing different types of sparsity which have been proposed since (<xref ref-type="bibr" rid="B41">Shen et al., 2013</xref>; <xref ref-type="bibr" rid="B24">Kim et al., 2017</xref>). Finally, <italic>PINSPlus</italic> (<xref ref-type="bibr" rid="B35">Nguyen et al., 2018</xref>), to identify subtypes across different views, uses a perturbation scheme applied to each source of data to define stable clusters, before merging results using different algorithms to construct a similarity matrix based on the overall connectivity of the patients.</p>
</sec>
<sec id="s2-2-3">
<title>2.2.3 Limitations</title>
<p>Concomitant approaches can be suitable for solving problems related to clinical and genomic datasets. However, none of these approaches provides an explicit recipe for dealing with heterogeneous data.<xref ref-type="fn" rid="fn2">
<sup>1</sup>
</xref> In particular, the longitudinal aspect is not taken into account in these kinds of approaches. In addition, most methods require new representations derived from the original space. Distorting the initial information may significantly complicate the posterior validation of the extracted features. The inherent limitation of methods based on dimensionality reduction was referred to previously. An additional difficulty arises with methods based on similarity matrices, such as kernel methods that implicitly map the data in a new feature space, since these methods require a pre-image problem to be solved for features to be approximated and, where possible, interpreted.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 Mixtures of Regressions With Clinical and Genomic Data</title>
<p>To take advantage of both the clinical and the genomic information, the two datasets can be used simultaneously <italic>via</italic> a mixture model. Mixtures of experts provide an elegant framework for including concomitant variables as secondary information alongside subtype data (<xref ref-type="bibr" rid="B15">Gormley et al., 2019</xref>). This section starts with a description of mixture-of-experts models, with a view to clarify the links between this framework and the approach that we are proposing.</p>
<sec id="s3-1">
<title>3.1 Mixture-of-Experts Models</title>
<p>Let <bold>Y</bold> be a matrix of <italic>N</italic> observed outcomes represented by variables <italic>v</italic> &#x2208; {1&#x22ef;<italic>V</italic>} such that <bold>y</bold>
<sub>
<italic>i</italic>
</sub> &#x3d; (<italic>y</italic>
<sub>
<italic>i</italic>1</sub>, &#x2026; , <italic>y</italic>
<sub>
<italic>iv</italic>
</sub>, &#x2026; , <italic>y</italic>
<sub>
<italic>iV</italic>
</sub>), for <italic>i</italic> &#x2208; {1&#x22ef;<italic>N</italic>}. These observations come from a population of <italic>K</italic> components. <bold>z</bold> &#x3d; (<italic>z</italic>
<sub>1</sub>, &#x2026; , <italic>z</italic>
<sub>
<italic>i</italic>
</sub>, &#x2026; , <italic>z</italic>
<sub>
<italic>N</italic>
</sub>) is the component membership vector where <italic>z</italic>
<sub>
<italic>i</italic>
</sub> &#x2208; {1&#x22ef;<italic>K</italic>}, and <bold>Z</bold> is the corresponding indicator matrix such that <bold>z</bold>
<sub>
<italic>i</italic>
</sub> &#x2208; {0,1}<sup>
<italic>K</italic>
</sup>, with <italic>z</italic>
<sub>
<italic>ik</italic>
</sub> &#x3d; 1 if the observation <italic>i</italic> belongs to the <italic>k</italic>
<sup>th</sup> component and <italic>z</italic>
<sub>
<italic>ik</italic>&#x2032;</sub> &#x3d; 0, otherwise (<italic>&#x2200;k</italic>&#x2032; &#x2260; <italic>k</italic>). A matrix <bold>G</bold> of <italic>N</italic> concomitant data represented by variables <italic>&#x2113;</italic> &#x2208; {1&#x22ef;<italic>L</italic>} is also available, with <bold>g</bold>
<sub>
<italic>i</italic>
</sub> &#x3d; (<italic>g</italic>
<sub>
<italic>i</italic>1</sub>, &#x2026; , <italic>g</italic>
<sub>
<italic>i&#x2113;</italic>
</sub>, &#x2026; , <italic>g</italic>
<sub>
<italic>iL</italic>
</sub>), for <italic>i</italic> &#x2208; {1&#x22ef;<italic>N</italic>}. The random vectors corresponding to these representations are respectively denoted by Y, Z, and G.</p>
<p>
<italic>Remark.</italic> To lighten notations, the range of indexes will often be omitted, in which case the ranges of indexes <italic>i</italic>, <italic>v</italic>, <italic>&#x2113;,</italic> and <italic>k</italic> (or <italic>k</italic>&#x2032;) will be as defined previously.</p>
<p>Using the terminology in Gormley (21, Section 2.3), we are interested in <italic>simple mixture-of-experts models</italic> where the outcome data distribution depends on the latent component membership, which itself depends on the concomitant variables, such that <inline-formula id="inf1">
<mml:math id="m1">
<mml:mi mathvariant="double-struck">P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.3333em"/>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mspace width="0.3333em"/>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, with<disp-formula id="e1a">
<mml:math id="m2">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="0.3333em"/>
<mml:mo>,</mml:mo>
</mml:math>
<label>(1a)</label>
</disp-formula>
<disp-formula id="e1b">
<mml:math id="m3">
<mml:mtext>and</mml:mtext>
<mml:mspace width="2em"/>
<mml:mi mathvariant="double-struck">P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(1b)</label>
</disp-formula>where &#x398;<sub>
<italic>k</italic>
</sub>(&#x22c5;) is the set of parameters of the <italic>k</italic>
<sup>th</sup> component density function <italic>f</italic>
<sub>
<italic>k</italic>
</sub>(&#x22c5;; &#x398;<sub>
<italic>k</italic>
</sub>(&#x22c5;)), that is, the <italic>k</italic>
<sup>th</sup> expert, and <italic>&#x3b7;</italic>
<sub>
<italic>k</italic>
</sub>(&#x22c5;) the probability weight related to the <italic>k</italic>
<sup>th</sup> expert.</p>
</sec>
<sec id="s3-2">
<title>3.2 Proposed Approach</title>
<p>Based on the previously described framework, we propose a mixture-of-regressions model over time for disease subtyping, where patient symptoms are recorded from their follow-up along with genetic markers as concomitant variables. Each cluster thus describes the evaluation of the symptoms over time and is simultaneously linked to a set of genetic markers.</p>
<sec id="s3-2-1">
<title>3.2.1 Specificity</title>
<p>Our model is designed to take into account the longitudinal aspect of the clinical data and the high-dimensional nature of the genetic data. <bold>Y</bold> comprises observed values of clinical variables over a series of follow-up visits indexed by <italic>j</italic>. The <italic>v</italic>th clinical variable observed during the <italic>j</italic>
<sup>th</sup> visit of patient <italic>i</italic> is denoted <inline-formula id="inf2">
<mml:math id="m4">
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>. Also, the number of variables in the genetic data <bold>G</bold> may be of the order of a few million after genotype imputation, so that dedicated metrics [such as CADD (<xref ref-type="bibr" rid="B37">Rentzsch et al., 2018</xref>), used in our explanation concerning Parkinson&#x2019;s disease] or more general elimination techniques such as screening rules [see (<xref ref-type="bibr" rid="B34">Ndiaye et al., 2017</xref>) for instance] may still be required beforehand. Note that even where this kind of prior processing occurs, we remain in a configuration where <italic>N</italic> &#x226a; <italic>L</italic>.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Model</title>
<p>To connect our proposal with the mixture of experts given previously in <xref ref-type="disp-formula" rid="e1a">Eqs 1a</xref>, <xref ref-type="disp-formula" rid="e1b">1b</xref>, we characterize the problem as<disp-formula id="e2a">
<mml:math id="m5">
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mspace width="0.3333em"/>
<mml:mo>,</mml:mo>
</mml:math>
<label>(2a)</label>
</disp-formula>
<disp-formula id="e2b">
<mml:math id="m6">
<mml:mtext>and</mml:mtext>
<mml:mspace width="2em"/>
<mml:mi mathvariant="double-struck">P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(2b)</label>
</disp-formula>defining the following regression model with logistic weights:<disp-formula id="e3a">
<mml:math id="m7">
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mo>&#x3d;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">vkp</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.3333em"/>
<mml:msubsup>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mspace width="0.17em"/>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:math>
<label>(3a)</label>
</disp-formula>
<disp-formula id="e3b">
<mml:math id="m8">
<mml:mtext>such&#x2009;that</mml:mtext>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:munderover accentunder="false" accent="false">
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mo>&#x3d;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">vkp</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.3333em"/>
<mml:msubsup>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mspace width="0.17em"/>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(3b)</label>
</disp-formula>
<disp-formula id="e3c">
<mml:math id="m9">
<mml:mtext>and</mml:mtext>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x22ba;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:munder>
<mml:mi>exp</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x22ba;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:math>
<label>(3c)</label>
</disp-formula>where<list list-type="simple">
<list-item>
<p>&#x2022; <italic>t</italic>
<sub>
<italic>ij</italic>
</sub> is the time metric, which might, for example, be the patient&#x2019;s age or time since the disease was first diagnosed, for the patient <italic>i</italic> at their <italic>j</italic>th follow-up visit,</p>
</list-item>
<list-item>
<p>&#x2022; <italic>p</italic> &#x2208; {0&#x22ef;<italic>P</italic>} is the polynomial degree considered in the regression (<italic>P</italic> &#x3d; 2 is generally sufficient),</p>
</list-item>
<list-item>
<p>&#x2022; {<italic>&#x3b1;</italic>
<sub>
<italic>vkp</italic>
</sub>}, {<italic>&#x3c3;</italic>
<sub>
<italic>vk</italic>
</sub>}, and {<bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>k</italic>
</sub>} are parameters or vectors to be estimated, with {<italic>&#x3c9;</italic>
<sub>1<italic>&#x2113;</italic>
</sub>} &#x3d; 0 for the sake of identifiability,</p>
</list-item>
<list-item>
<p>&#x2022; <inline-formula id="inf3">
<mml:math id="m10">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:munder>
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>, implies some conditional independence assumptions between variables, patients, and visits when the class is known. The clinical variables are chosen to be as independent as possible, the correlation between individuals should essentially come from a similar typology of the disease, and finally, the remaining time correlation after the polynomial regression is expected to be poor. If the Gaussian hypothesis does not apply to the variable <italic>v</italic>, Poisson or logistic regression may be considered instead, with no substantial additional cost.</p>
</list-item>
</list>
</p>
<p>The longitudinal aspect is taken into account by assuming for each cluster the existence of typical temporal trajectories, described by a polynomial regression of clinical variables over time, around which the patients&#x2019; symptoms evolve. These are assumed to fully summarize the temporal evolution of each patient. According to this model, there is no residual intra-patient correlation conditional on the trajectory followed (requiring knowledge of the cluster). The modeling of posterior probabilities <italic>via</italic> logistic regression allows concomitant variables, such as genetic data, to subtly influence the subtyping.</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Model Selection</title>
<p>We combine two model selection strategies to select the hyperparameters involved in the mixture. The first of these is the Bayesian Information Criterion (BIC), which is widely used within the research community to select <italic>K</italic>, the most appropriate number of subtypes, and <italic>P</italic>, the polynomial degrees in the main regressions. Also, as discussed previously, we suspect that many variables <italic>&#x2113;</italic> from <bold>G</bold> will have little or no influence on disease phenomenology. A Lasso penalization is therefore applied on the coefficients {<bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>k</italic>
</sub>}, <italic>&#x2200;k</italic>, to select those that have the most relevance in the subtyping. More details about this aspect are given in <xref ref-type="sec" rid="s4">Section 4</xref>.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Expectation-Maximization Algorithm With Integrated Lasso Inference</title>
<p>The inference of this kind of model with latent variables, here {<italic>z</italic>
<sub>
<italic>ik</italic>
</sub>}, is traditionally done with the aid of an Expectation-Maximization algorithm [EM algorithm, (<xref ref-type="bibr" rid="B7">Dempster et al., 1977</xref>)]. We use a modified version of this algorithm with a Lasso-type penalized likelihood instead of classical likelihood.</p>
<sec id="s4-1">
<title>4.1 Expectation-Maximization Algorithm</title>
<p>The (<italic>q</italic> &#x2b; 1)th iteration of the modified EM algorithm maximizes the expected and penalized complete-data log-likelihood <inline-formula id="inf4">
<mml:math id="m11">
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">Y</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mi mathvariant="bold">G</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold">Z</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">&#x3c3;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="script">P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> which reads<disp-formula id="equ1">
<mml:math id="m12">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>log</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:math>
</disp-formula>where <italic>&#x3bb;</italic> &#x3e; 0 controls the amount of sparsity applied on the <italic>&#x2113;</italic>
<sub>1</sub> norm of <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>k</italic>
</sub> and where <italic>&#x3b7;</italic>
<sub>
<italic>k</italic>
</sub>(&#x22c5;; &#x22c5;) and <italic>f</italic>
<sub>
<italic>k</italic>
</sub>(&#x22c5;; &#x22c5;) are defined as in <xref ref-type="disp-formula" rid="e3a">Eqs 3a</xref>,<xref ref-type="disp-formula" rid="e3b">3b</xref>,<xref ref-type="disp-formula" rid="e3c">3c</xref>.</p>
<p>To maximize the expected and penalized complete-data log-likelihood, each iteration is separated into an expectation step (E) followed by a maximization step (M).<list list-type="simple">
<list-item>
<p>&#x2022; At step E of the (<italic>q</italic> &#x2b; 1)th iteration, posterior weights are updated as follows:</p>
</list-item>
</list>
<disp-formula id="equ2">
<mml:math id="m13">
<mml:mtable class="align-star" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi mathvariant="normal">Y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:munder>
<mml:mrow>
<mml:mo>&#x220f;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mo>&#x220f;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:munder>
<mml:mrow>
<mml:mo>&#x220f;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mo>&#x220f;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<list list-type="simple">
<list-item>
<p>&#x2022; At step M of the (<italic>q</italic> &#x2b; 1)th iteration, parameters are updated as follows:</p>
</list-item>
</list>
<disp-formula id="equ3">
<mml:math id="m14">
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">g</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">&#x398;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>log</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:math>
</disp-formula>
</p>
<p>The maximization with regard to parameters {<bold>
<italic>&#x3b1;</italic>
</bold>, <bold>
<italic>&#x3c3;</italic>
</bold>} presents no difficulty (<xref ref-type="sec" rid="s12">Supplementary Material</xref>). However, there is no closed formula that may be used for updating the logistic weights parameters. The term to be maximized with respect to {<bold>
<italic>&#x3c9;</italic>
</bold>} at iteration (<italic>q</italic> &#x2b; 1) of the EM algorithm is<disp-formula id="e4">
<mml:math id="m15">
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">g</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="0.17em"/>
<mml:mo>;</mml:mo>
<mml:mspace width="0.17em"/>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c9;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>This maximization problem corresponds to the multinomial logistic regression problem with a <italic>&#x2113;</italic>
<sub>1</sub> penalty, which can be solved using a proximal-Newton approach (<xref ref-type="bibr" rid="B19">Hastie et al., 2015</xref>). <xref ref-type="fn" rid="fn3">
<sup>2</sup>
</xref>
</p>
</sec>
<sec id="s4-2">
<title>4.2 Initialization and Variable Selection in Practice</title>
<p>The EM algorithm is subject to local optima. To address this classical problem and to provide stability and improve robustness, we perform a variety of initializations and retain the initialization that yields the lowest BIC.</p>
<p>Strategies commonly used for selecting the hyperparameter <italic>&#x3bb;</italic> are based on adjusted information criterion [ (<xref ref-type="bibr" rid="B4">Chen and Chen, 2012</xref>) for General Linear Models or (<xref ref-type="bibr" rid="B9">Fop and Murphy, 2018</xref>) for a more general overview]. In an original approach, <xref ref-type="bibr" rid="B47">Yi and Caramanis (2015</xref>) proposed optimizing the hyperparameter <italic>&#x3bb;</italic> <italic>via</italic> an iterative scheme over successive M steps, and showed local convergence properties in the high dimensional setting.</p>
<p>In this work, we use an alternative adopted by <xref ref-type="bibr" rid="B33">Mortier et al. (2015</xref>), where <italic>&#x3bb;</italic> is chosen within the M step by cross-validation such that the likelihood of the multinomial logistic model (4) is maximized. The simulation study described in <xref ref-type="sec" rid="s5">Section 5</xref> showed that proceeding with this selection at every M step of the EM algorithm does not compromise convergence.</p>
<p>Finally, to avoid (negative) bias due to the penalization in the parameter estimation, we re-estimate the selected {<bold>
<italic>&#x3c9;</italic>
</bold>} parameters at the end of the EM algorithm to obtain the maximum likelihood estimates, which is the usual practice [(<xref ref-type="bibr" rid="B18">Hastie et al., 2009</xref>), p. 91].</p>
</sec>
<sec id="s4-3">
<title>4.3 Implementation</title>
<p>The implementation of the method proposed in this study, which we have named <monospace>DiSuGen</monospace>, and an <monospace>R</monospace> Markdown tutorial are available for download (<xref ref-type="bibr" rid="B6">Courbariaux et al., 2020</xref>). We build on the <monospace>FlexMix R</monospace> package (<xref ref-type="bibr" rid="B16">Grun and Leisch, 2008</xref>) which proposes an EM algorithm suitable for multinomial logistic mixture models. To implement our method, we developed an adapted concomitant variable driver making use of <monospace>glmnet</monospace> within <monospace>FlexMix</monospace>. For a faster convergence, we resort in practice to a Classification EM (CEM) algorithm in which <inline-formula id="inf5">
<mml:math id="m16">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> are replaced by the indicator variables <inline-formula id="inf6">
<mml:math id="m17">
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B3">Celeux and Govaert, 1992</xref>).</p>
</sec>
</sec>
<sec id="s5">
<title>5 Numerical Tests Using Artificial Data</title>
<p>We used artificial data to test our proposed estimation and model selection procedures. These simulations are designed to assess the ability of the CEM algorithm both to produce a good estimation of the parameters and to obtain the appropriate model. The methodology is given in detail as follows.</p>
<sec id="s5-1">
<title>5.1 Data Generation</title>
<p>Artificial data are simulated according to the model (3) with <italic>N</italic> &#x3d; 396 patients, <italic>V</italic> &#x3d; 4 clinical variables, <italic>K</italic> &#x3d; 3 clusters, <italic>P</italic> &#x3d; 1 polynomial degree in the regression, three follow-up visits per patient with times <italic>t</italic>
<sub>
<italic>ij</italic>
</sub> randomly ranging from 10 to 410 days for the first visit, from 1,800 to 2,200&#xa0;days for the second, and from 3,600 to 4,000&#xa0;days for the third. Also, <italic>L</italic> &#x3d; 2,657 genetic markers are simulated with only 10 having an influence on the clustering such that <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>k</italic>{<italic>&#x2113;</italic>}</sub> is <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>2{2,3,4}</sub> &#x3d; <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>3{5,6,7}</sub> &#x3d; 2, <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>2{5,6,7}</sub> &#x3d; &#x2212;1, and <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>3{1,8,9,10}</sub> &#x3d; &#x2212;2. For the sake of consistency with the study presented in <xref ref-type="sec" rid="s6">Section 6</xref>, the genetic markers come from the Parkinson&#x2019;s disease genetic data, and the parameters {<bold>
<italic>&#x3b1;</italic>
</bold>, <bold>
<italic>&#x3c3;</italic>
</bold>} are chosen to be realistic with regard to the Parkinson&#x2019;s disease clinical data.</p>
</sec>
<sec id="s5-2">
<title>5.2 Protocol</title>
<p>For each simulation, the proposed CEM algorithm is run with <italic>K</italic> &#x3d; 3 clusters and a Lasso penalty. The estimation is initialized with 10 sets of starting values corresponding to 10 random assignments into <italic>K</italic> &#x3d; 3 clusters, and the set of values that gives the lowest BIC is retained. The experiment is repeated 100 times. To assess the performance of our method, we compare many different methods:<list list-type="simple">
<list-item>
<p>&#x2022; The <italic>integrative</italic> method is the one described in this study, which uses both clinical and genetic data and estimates the parameters {<bold>
<italic>&#x3b1;</italic>
</bold>, <bold>
<italic>&#x3c3;</italic>
</bold>} and {<bold>
<italic>&#x3c9;</italic>
</bold>}, and the subtypes <bold>z</bold>.</p>
</list-item>
<list-item>
<p>&#x2022; The <italic>oracle integrative</italic> and <italic>semi-oracle integrative</italic> methods also use both clinical and genetic data to estimate the subtypes <bold>z</bold>. The <italic>oracle integrative</italic> uses all parameters of the model set to their true values. The <italic>semi-oracle integrative</italic> method sets only the parameters {<bold>
<italic>&#x3c9;</italic>
</bold>} to their true values. This allows us to check to what extent our method correctly subtypes the data and estimates the parameters relating to clinical variables (semi-oracle) and genetic variables (oracle).</p>
</list-item>
<list-item>
<p>&#x2022; The <italic>2-step</italic> method does not use genetic information in the clustering process to estimate the parameters {<bold>
<italic>&#x3b1;</italic>
</bold>, <bold>
<italic>&#x3c3;</italic>
</bold>}, and clusters have constant weights, that is, <inline-formula id="inf7">
<mml:math id="m18">
<mml:mi mathvariant="double-struck">P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>. In this case, the Lasso-penalized multinomial logistic regression is performed afterward to give genetic association results. This allows us to assess the benefit of including genetic information in the clustering process at the same time that the clinical parameters are estimated.</p>
</list-item>
<list-item>
<p>&#x2022; The <italic>oracle 2-step</italic> method is identical to the <italic>2-step</italic> method, except that the parameters {<bold>
<italic>&#x3b1;</italic>
</bold>, <bold>
<italic>&#x3c3;</italic>
</bold>} are set to their true values.</p>
</list-item>
<list-item>
<p>&#x2022; Where possible, the proposed method is also compared with the <italic>K</italic>-means method, which corresponds to a simple Gaussian mixture model with identical proportions and identical standard deviations in all clusters. For this purpose, we use a <italic>K</italic>-means method adapted to longitudinal data implemented in the <monospace>R</monospace> package <monospace>kml3d</monospace> (<xref ref-type="bibr" rid="B12">Genolini et al., 2015</xref>).</p>
</list-item>
</list>
</p>
</sec>
<sec id="s5-3">
<title>5.3 Results</title>
<sec id="s5-3-1">
<title>5.3.1 Clustering Ability</title>
<p>The Adjusted Rand Index [ARI, (<xref ref-type="bibr" rid="B36">Rand, 1971</xref>; <xref ref-type="bibr" rid="B22">Hubert and Arabie, 1985</xref>)] is computed for each simulation to check that the estimated clusters are close to those that are being simulated (a higher ARI score is more desirable). The results for our proposed method are shown in the boxplot for the <italic>integrative</italic> method in <xref ref-type="fig" rid="F1">Figure 1</xref>. Most clusters are well identified. Where no use is made of genetic information within the clustering, ARI values obtained by the algorithm are lower. As we might expect, the algorithm making use of genetic data achieves better clustering, as shown by the <italic>oracle integrative</italic> results. This improvement in cluster prediction is not, however, because of a better estimation of the parameters {<bold>
<italic>&#x3c9;</italic>
</bold>}, as shown by the <italic>semi-oracle integrative</italic> results. Finally, the <italic>K</italic>-means algorithm is less effective in recovering the underlying classification, with ARI values all between 0.6 and 0.7. This was expected, since the differences between the clusters partly lie in the variances of the variables. Moreover, the <italic>K</italic>-means method does not address the times of the follow-up visits, but only their sequence numbers.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Results on artificial data over 100 simulations. ARI with a <italic>K</italic>-means algorithm, the <italic>2-step</italic> method (no use of genetic information) and the integrative method (use of genetic information) and their corresponding oracles. The <italic>y</italic>-axis represents the ARI score (the higher the better).</p>
</caption>
<graphic xlink:href="fgene-13-859462-g001.tif"/>
</fig>
</sec>
<sec id="s5-3-2">
<title>5.3.2 Parameter Estimation Ability</title>
<p>The parameters of the main regressions are estimated accurately and with biases close to 0 irrespective of the clinical variable considered and the approach used (<italic>2-step</italic> or <italic>integrative</italic>). Taking genetic information into account does not appear to offer any great improvement in the estimation of these parameters. Regarding the logistic regression parameters, the sign of the estimated parameters is mostly reflected correctly in the two approaches, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Results on artificial data over 100 simulations. Non-negative {<bold>
<italic>&#x3c9;</italic>
</bold>} parameter estimates and their respective true values. The <italic>y</italic>-axis represents the values of the estimates.</p>
</caption>
<graphic xlink:href="fgene-13-859462-g002.tif"/>
</fig>
</sec>
<sec id="s5-3-3">
<title>5.3.3 Variable Selection Within the Logistic Regression</title>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> summarizes the results of the proposed Lasso selection procedure with regard to the genetic variables. The sensitivity of the proposed <italic>integrative</italic> approach (52.7%) is higher than that of the equivalent <italic>2-step</italic> method (46.8%). It was computed globally over the 100 simulations and for the 10 active genetic variables. With both methods, the selection rates of 8 of the 10 active genetic variables are notably higher than the selection rates of the other variables, which indicates that the selection method performs well. The two remaining active markers do not vary between patients, and could therefore be replaced by any variant with a low variation. The specificity of both approaches is good, with a slightly better result for the <italic>2-step</italic> method (98.9%, vs. 98.2% for the <italic>integrative</italic> approach). This result is an overall result computed over the 100 simulations and for the 2,647 inactive genetic variables. Selection performance decreases, as expected, the closer the parameters {<bold>
<italic>&#x3c9;</italic>
</bold>} are set to zero (data not shown).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Global sensitivity and specificity of the integrative method compared with the 2-step method for the selection of the genetic variables in the artificial data experiment with 100 simulations. Among 2,657 variables, 10 had to be selected. The number of times these variables have been selected over the 100 simulations is also specified for both methods.</p>
</caption>
<graphic xlink:href="fgene-13-859462-g003.tif"/>
</fig>
</sec>
<sec id="s5-3-4">
<title>5.3.4 Selection Ability of the Model</title>
<p>An additional simulation was done to evaluate the capacity of the BIC (computed as described in <xref ref-type="sec" rid="s3">Section 3</xref>) to select the correct number of clusters (<italic>K</italic> &#x3d; 3) on the same 100 simulated datasets. The results are shown as the histogram in <xref ref-type="fig" rid="F4">Figure 4</xref>. The correct number of clusters is selected 79 times out of 100.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Results on artificial data over 100 simulations. Estimated number of clusters according to the BIC. The <italic>y</italic>-axis represents the number of simulations for the number of clusters selected in the <italic>x</italic>-axis.</p>
</caption>
<graphic xlink:href="fgene-13-859462-g004.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s6">
<title>6 Demonstration Using Parkinson&#x2019;s Disease Subtyping</title>
<p>We applied our proposed method to PD subtyping. PD is known to have several subtypes, and there are a number of relevant studies, including the study by <xref ref-type="bibr" rid="B27">Lewis et al. (2005</xref>).</p>
<sec id="s6-1">
<title>6.1 Data Description</title>
<p>The data on which we applied our method come from the DIG-PD cohort (<xref ref-type="bibr" rid="B5">Corvol et al., 2018</xref>) comprising 396 genotyped adults with a recent PD onset (diagnosed less than 6&#xa0;years before the beginning of the study).</p>
<sec id="s6-1-1">
<title>6.1.1 Clinical Data</title>
<p>Clinical data were collected at inclusion and then at yearly clinical follow-ups between 1 and 7&#xa0;years. They include scores evaluating the progression of the disease. Two of these scores are taken to be representatives of the evolution of the disease, namely <italic>UPDRS III</italic> (Section III of the Unified PD Rating Scale, a motor examination), and <italic>MMSE</italic> (the score from the Mini-Mental Status Examination tool kit, an evaluation of cognitive impairment). The higher UPDRS III and the lower MMSE, the greater the degree of impairment will be. These two scores were adjusted beforehand for gender effects and for treatment doses by considering the residuals of the linear regression with gender and treatment doses as (factor and quantitative, respectively) predictors. The time scale used is patient age.</p>
</sec>
<sec id="s6-1-2">
<title>6.1.2 Genetic Data</title>
<p>More than six million genetic markers were available after imputation for each patient. Only 2,652 of them were used, namely, those that have been associated with PD in previous studies (about 400) together with those that have an important impact on gene function (scaled CADD score <xref ref-type="fn" rid="fn4">
<sup>3</sup>
</xref> greater than 25) and an allele frequency greater than 0.01. As done classically, genetic markers with two copies of the reference allele were encoded &#x2212;1, those with two copies of the alternative allele were encoded 1, and the remainder (with one copy of each) were encoded 0.</p>
</sec>
</sec>
<sec id="s6-2">
<title>6.2 Results</title>
<sec id="s6-2-1">
<title>6.2.1 Model Selection Results</title>
<p>To ensure good interpretability of results, the number of clusters was limited to <italic>K</italic> &#x3d; 4, and no more than two polynomial degrees were tested. The solution with the lowest BIC was obtained with four clusters and one polynomial degree.</p>
</sec>
<sec id="s6-2-2">
<title>6.2.2 Clinical Results</title>
<p>Clustering results obtained from the clinical data are shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. Note that the variables shown here are residuals of a fitted linear model adjusted with gender and treatment doses. Patients are allocated to the cluster to which they are most likely to belong according to the model. Half of the patients are allocated to cluster 1 and the other half is allocated to three remaining clusters in approximately equal measure. The four clusters correspond to different ways in which the disease may evolve: low motor scores and no cognitive evolution (cluster 1, a mild form of PD), high motor scores and no cognitive evolution (cluster 2, a more severe motor form), high motor scores and significant cognitive evolution (clusters 3 and 4, a severe form and an intermediate form of PD). Moreover, the cluster structure is significantly related to the age of diagnosis which was not used in the clustering process. In particular, cluster 4 shows clear signs of diagnosis at a later age (<xref ref-type="fig" rid="F6">Figure 6</xref>).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Clustering with regard to the clinical variables. The top and bottom left-hand graphs show the fitted trajectories (straight lines) for each of the four clusters and the corresponding 68% confidence intervals obtained by adding and subtracting the fitted <italic>&#x3c3;</italic> parameters. The top <italic>y</italic>-axis represents the MMSE score (evaluation of cognitive impairment) and the bottom <italic>y</italic>-axis the UPDRS III score (motor evaluation). The other graphs show in detail, for each cluster and each score around the standard trajectory, all the trajectories of the patients assigned to the cluster.</p>
</caption>
<graphic xlink:href="fgene-13-859462-g005.tif"/>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Boxplot of the age at diagnosis for each of the clusters.</p>
</caption>
<graphic xlink:href="fgene-13-859462-g006.tif"/>
</fig>
</sec>
<sec id="s6-2-3">
<title>6.2.3 Genetic Association Results</title>
<p>
<xref ref-type="fig" rid="F7">Figure 7</xref> shows the results with 95% confidence intervals linked to the parameters {<bold>
<italic>&#x3c9;</italic>
</bold>} <xref ref-type="fn" rid="fn5">
<sup>4</sup>
</xref>. The <italic>p</italic>-values below 0.05 (i.e., significant association before any multiple test correction) correspond to a 0 value outside the 95% confidence interval.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Genetic association: estimated logistic regression parameters {<bold>
<italic>&#x3c9;</italic>
</bold>}. Cluster 1 is the reference: <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>&#x2113;</italic>&#x3d;1</sub> &#x3d; <bold>0</bold>; top <bold>(A)</bold>: <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>&#x2113;</italic>&#x3d;2</sub>; middle <bold>(B)</bold>: <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>&#x2113;</italic>&#x3d;3</sub>; and bottom <bold>(C)</bold>: <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>&#x2113;</italic>&#x3d;4</sub>. The confidence intervals are computed from the Hessian matrix provided by the <monospace>R</monospace> function <monospace>nnet::nnet</monospace> (<xref ref-type="bibr" rid="B38">Ripley et al., 2016</xref>).</p>
</caption>
<graphic xlink:href="fgene-13-859462-g007.tif"/>
</fig>
<p>There were 15 SNPs selected, 7 of which belong to genes that potentially have a role in neurological diseases. Among the selected SNPs, rs35866326 (which appears in the <bold>
<italic>&#x3c9;</italic>
</bold>
<sub>
<italic>&#x2113;</italic>&#x3d;2</sub> panel of <xref ref-type="fig" rid="F7">Figure 7</xref>) has been a focus of attention in the PD literature, having been associated with susceptibility to PD (<xref ref-type="bibr" rid="B31">Maraganore et al., 2005</xref>; <xref ref-type="bibr" rid="B14">Goris et al., 2006</xref>; <xref ref-type="bibr" rid="B30">Maraganore et al., 2006</xref>) although other studies (<xref ref-type="bibr" rid="B8">Farrer et al., 2006</xref>; <xref ref-type="bibr" rid="B28">Li et al., 2006</xref>) have failed to replicate this result. The lack of consensus might be because of this gene&#x2019;s association only with a particular subtype of PD, as suggested in the present study, where it is associated with cluster 2 only. However, an unselected variant does not rule out any association with the disease subtype. It may be associated, but not sufficiently to contribute more information relative to the clustering.</p>
</sec>
</sec>
</sec>
<sec id="s7">
<title>7 Conclusion</title>
<sec id="s7-1">
<title>7.1 Synthesis and Results</title>
<p>We proposed a model-based method for disease subtyping where the information comes from both short longitudinal data with varying observation times, as clinical follow-up data often are, and from high-dimensional quantitative data, such as genotyping data. Unlike in most multi-view clustering methods, the data are processed in a non-symmetrical way by integrating genetic data in the clustering <italic>via</italic> multinomial logistic weights. A Lasso penalty on the logistic regression parameters addresses the high-dimensionality of the genotyping data while exhibiting a short list of genetic factors potentially involved in the typology of the disease.</p>
<p>An experiment on artificial data validates our proposed inference and model selection approach and shows that it is better able to identify latent subtypes of the disease and influential genetic factors than an approach that first clusters clinical data and then performs an association study. When our method is applied on clinical and genetic data from a cohort of patients with Parkinson&#x2019;s disease, we are able to characterize four distinct subtypes and 15 genetic factors with a potential impact on subtyping. Of these 15 SNPs, the most significant SNP is already associated with PD. Half of the others belong to genes suspected to be involved in neurological diseases. Being able to recover results like these shows the relevance of our approach in a real setting.</p>
</sec>
<sec id="s7-2">
<title>7.2 Perspectives</title>
<p>Several aspects might be revisited in future works, as outlined as follows.</p>
<sec id="s7-2-1">
<title>7.2.1 Replication</title>
<p>The statistical analysis presented here uses a relatively small sample size and it may thus be of interest to attempt to replicate and confirm our results using independent cohorts.</p>
</sec>
<sec id="s7-2-2">
<title>7.2.2 Modeling of Data</title>
<p>If the objective of the subtyping is to predict the evolution of the patient&#x2019;s symptoms, and if more data are available for each patient, then the temporal dynamics specific to each individual might be addressed in a more refined way, for example, using a Gaussian process as done by <xref ref-type="bibr" rid="B40">Schulam and Saria (2015</xref>). In addition, if the focus is on correlated clinical variables, a multivariate version of the proposed model would be interesting, but this is complicated by the functional nature of the data (<italic>t</italic>
<sub>
<italic>ij</italic>
</sub> times are different from one individual <italic>i</italic> to another). Regarding the genetic data, a lighter preprocessing step for the purposes of elimination may be desirable in a very high-dimensional setting (with several million SNPs), and it may consequently be useful to summarize the data, for instance, by aggregating SNPs in linkage disequilibrium blocks (<xref ref-type="bibr" rid="B17">Guinot et al., 2018</xref>).</p>
</sec>
<sec id="s7-2-3">
<title>7.2.3 Association Study With Genetic Data</title>
<p>Finally, our proposed method does not dispense the need for a more traditional association study afterward, and this presents an opportunity for studying further potential associations between the genetic markers extracted in the variable selection process.</p>
<p>To this end, a correction for multiple testing might be done to assess the likelihood that the SNPs identified with our method actually have an impact on the disease typology. This correction should take into account the fact that the Lasso selection is performed on a large number of SNPs and that the tests are performed on a subgroup of those SNPs. Post-hoc inference tests may, therefore, be useful (<xref ref-type="bibr" rid="B13">Goeman and Solari, 2011</xref>).</p>
</sec>
</sec>
</sec>
</body>
<back>
<sec id="s8">
<title>Data Availability Statement</title>
<p>The data analyzed in this study are subject to the following licenses/restrictions: the datasets analyzed for this study belong to the APHP (Assistance Publique H&#xf4;pitaux de Paris), and can be made available upon request from J-CC. Requests to access these datasets should be directed to J-CC, <email>jean-christophe.corvol@aphp.fr</email>.</p>
</sec>
<sec id="s9">
<title>Author Contributions</title>
<p>MC contributed to the design of the algorithm, the development of the <monospace>R</monospace> code, did the experiments, and wrote most of the manuscript. KD contributed to the development of the <monospace>R</monospace> code. CD contributed to the design of the algorithm and the writing of the paper. MS contributed to the writing of the paper, the design of the algorithm, and coordinated the experiments. CA contributed to the design of the algorithm, contributed to the development of the <monospace>R</monospace> code, coordinated the experiments, and the writing of the paper. Regarding the application on Parkinson, J-CC, FD, and SB provided the data, FD filtered the genetic data and J-CC interpreted the results.</p>
</sec>
<sec id="s10">
<title>Funding</title>
<p>This work was carried out as part of the MeMoDeeP project funded by the ANR and led by MM. The data used here are from the DIGPD cohort sponsored by the Assistance Publique H&#xf4;pitaux de Paris and funded by the French Ministry of Health (PHRC AOM0810).</p>
</sec>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>The methodological reflections behind this work grew out of sustained exchanges with the members of this project, notably Pierre Neuvial. We would also like to thank Agathe Guilloux for her help and enriching discussions around this work, as well as Franck Samson for his work on the (intranet) web interface of the method. We would like to thank the DIGPD study group for collecting the data, and the patients who formed this cohort. The data have been adapted for this specific work by J-CC, SB, FD, Graziela Mangonne, Alexis Elbaz, and Fanny Artaud.</p>
</ack>
<sec id="s13">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fgene.2022.859462/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fgene.2022.859462/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Presentation1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn2">
<label>1</label>
<p>For instance, clinical data may be represented by numerical scores observed on different visits (of a continuous nature with a longitudinal aspect), while genomic data may be represented by Single Nucleotide Polymorphisms (SNPs, of a categorial nature without a longitudinal aspect).</p>
</fn>
<fn id="fn3">
<label>2</label>
<p>For instance, with the <monospace>R</monospace> package <monospace>glmnet</monospace> (<xref ref-type="bibr" rid="B10">Friedman et al., 2010</xref>).</p>
</fn>
<fn id="fn4">
<label>3</label>
<p>Combined Annotation Dependent Depletion score, this score evaluates the deleteriousness of variants in the human genome (<xref ref-type="bibr" rid="B37">Rentzsch et al., 2018</xref>).</p>
</fn>
<fn id="fn5">
<label>4</label>
<p>The confidence intervals are computed from the Hessian matrix provided by the <monospace>R</monospace> function <monospace>nnet::nnet</monospace> (<xref ref-type="bibr" rid="B38">Ripley et al., 2016</xref>).</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bolte</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Sabach</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Teboulle</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Proximal Alternating Linearized Minimization for Nonconvex and Nonsmooth Problems</article-title>. <source>Math. Program.</source> <volume>146</volume>, <fpage>459</fpage>&#x2013;<lpage>494</lpage>. <pub-id pub-id-type="doi">10.1007/s10107-013-0701-9</pub-id> </citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bush</surname>
<given-names>W. S.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Chapter 11: Genome-wide Association Studies</article-title>. <source>PLoS Comput. Biol.</source> <volume>8</volume>, <fpage>e1002822</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1002822</pub-id> </citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Celeux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Govaert</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>A Classification EM Algorithm for Clustering and Two Stochastic Versions</article-title>. <source>Comput. Statistics Data Analysis</source> <volume>14</volume>, <fpage>315</fpage>&#x2013;<lpage>332</lpage>. <pub-id pub-id-type="doi">10.1016/0167-9473(92)90042-e</pub-id> </citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Extended BIC for Small-N-Large-P Sparse GLM</article-title>. <source>Stat. Sin.</source> <volume>22</volume>, <fpage>555</fpage>&#x2013;<lpage>574</lpage>. <pub-id pub-id-type="doi">10.5705/ss.2010.216</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Corvol</surname>
<given-names>J.-C.</given-names>
</name>
<name>
<surname>Artaud</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cormier-Dequaire</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Rascol</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Durif</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Derkinderen</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Longitudinal Analysis of Impulse Control Disorders in Parkinson Disease</article-title>. <source>Neurology</source> <volume>91</volume>, <fpage>e189</fpage>&#x2013;<lpage>e201</lpage>. <pub-id pub-id-type="doi">10.1212/wnl.0000000000005816</pub-id> </citation>
</ref>
<ref id="B6">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Courbariaux</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ambroise</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dalmasso</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Szafranski</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>DiSuGen: Disease Subtyping with Integrated Genetic Association</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/MCour/DiSuGen">https://github.com/MCour/DiSuGen</ext-link>
</comment>. </citation>
</ref>
<ref id="B16">
<citation citation-type="journal"> <comment>[Dataset]</comment>
<person-group person-group-type="author">
<name>
<surname>Grun</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Leisch</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>FlexMix Version 2: Finite Mixtures with Concomitant Variables and Varying and Constant Parameters</article-title>. <source>J. Stat. Softw.</source> <volume>28</volume> (<issue>4</issue>), <fpage>1</fpage>&#x2013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.18637/jss.v028.i04</pub-id> </citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dempster</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Laird</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Rubin</surname>
<given-names>D. B.</given-names>
</name>
</person-group> (<year>1977</year>). <article-title>Maximum Likelihood from Incomplete Data via theEMAlgorithm</article-title>. <source>J. R. Stat. Soc. Ser. B Methodol.</source> <volume>39</volume>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1977.tb01600.x</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Farrer</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Haugarvoll</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ross</surname>
<given-names>O. A.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Milkovic</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>Cobb</surname>
<given-names>S. A.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>Genomewide Association, Parkinson Disease, and PARK10</article-title>. <source>Am. J. Hum. Genet.</source> <volume>78</volume>, <fpage>1084</fpage>&#x2013;<lpage>1088</lpage>. <pub-id pub-id-type="doi">10.1086/504728</pub-id> </citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fop</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Murphy</surname>
<given-names>T. B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Variable Selection Methods for Model-Based Clustering</article-title>. <source>Stat. Surv.</source> <volume>12</volume>, <fpage>18</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1214/18-ss119</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Regularization Paths for Generalized Linear Models via Coordinate Descent</article-title>. <source>J. Stat. Softw.</source> <volume>33</volume>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.18637/jss.v033.i01</pub-id> </citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Vasilakos</surname>
<given-names>A. V.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An Overview of Recent Multi-View Clustering</article-title>. <source>Neurocomputing</source> <volume>402</volume>, <fpage>148</fpage>&#x2013;<lpage>161</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2020.02.104</pub-id> </citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Genolini</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Alacoque</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sentenac</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Arnaud</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Kml and Kml3d: R Packages to Cluster Longitudinal Data</article-title>. <source>J. Stat. Softw.</source> <volume>65</volume>, <fpage>1</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.18637/jss.v065.i04</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goeman</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Solari</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Multiple Testing for Exploratory Research</article-title>. <source>Stat. Sci.</source> <volume>26</volume>, <fpage>584</fpage>&#x2013;<lpage>597</lpage>. <pub-id pub-id-type="doi">10.1214/11-sts356</pub-id> </citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goris</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Williams-Gray</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Foltynie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Compston</surname>
<given-names>D. A. S.</given-names>
</name>
<name>
<surname>Barker</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Sawcer</surname>
<given-names>S. J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>No Evidence for Association with Parkinson Disease for 13 Single-Nucleotide Polymorphisms Identified by Whole-Genome Association Screening</article-title>. <source>Am. J. Hum. Genet.</source> <volume>78</volume>, <fpage>1088</fpage>&#x2013;<lpage>1090</lpage>. <pub-id pub-id-type="doi">10.1086/504726</pub-id> </citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gormley</surname>
<given-names>I. C.</given-names>
</name>
<name>
<surname>Fr&#xfc;hwirth-Schnatter</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Mixture of Experts Models</article-title>,&#x201d; in <source>Handbook of Mixture Analysis</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Fr&#xfc;hwirth-Schnatter</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Celeux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Robert</surname>
<given-names>C. P.</given-names>
</name>
</person-group> (<publisher-loc>Boca Raton, Florida, USA</publisher-loc>: <publisher-name>CRC Press</publisher-name>), <fpage>271</fpage>&#x2013;<lpage>307</lpage>. <comment>chap. 12</comment>. <pub-id pub-id-type="doi">10.1201/9780429055911-12</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guinot</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Szafranski</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ambroise</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Samson</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Learning the Optimal Scale for GWAS through Hierarchical SNP Aggregation</article-title>. <source>BMC Bioinforma.</source> <volume>19</volume>, <fpage>459</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-018-2475-9</pub-id> </citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Friedman</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2009</year>). <source>The Elements of Statistical Learning: Data Mining, Inference, and Prediction</source>. <publisher-loc>Berlin, Germany</publisher-loc>: <publisher-name>Springer Science &#x26; Business Media</publisher-name>. </citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wainwright</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Statistical Learning with Sparsity</article-title>. <source>Monogr. statistics Appl. Probab.</source> <volume>143</volume>, <fpage>143</fpage>. <pub-id pub-id-type="doi">10.1201/b18401</pub-id> </citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hayes</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Overview of Statistical Methods for Genome-wide Association Studies (GWAS)</article-title>,&#x201d; in <source>Genome-wide Association Studies and Genomic Prediction</source> (<publisher-loc>Berlin, Germany</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>149</fpage>&#x2013;<lpage>169</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-62703-447-0_6</pub-id> </citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chaudhary</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Garmire</surname>
<given-names>L. X.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>More Is Better: Recent Progress in Multi-Omics Data Integration Methods</article-title>. <source>Front. Genet.</source> <volume>8</volume>, <fpage>84</fpage>. <pub-id pub-id-type="doi">10.3389/fgene.2017.00084</pub-id> </citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hubert</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Arabie</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1985</year>). <article-title>Comparing Partitions</article-title>. <source>J. Classif.</source> <volume>2</volume>, <fpage>193</fpage>&#x2013;<lpage>218</lpage>. <pub-id pub-id-type="doi">10.1007/bf01908075</pub-id> </citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jacques</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Preda</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Functional Data Clustering: a Survey</article-title>. <source>Adv. Data Anal. Classif.</source> <volume>8</volume>, <fpage>231</fpage>&#x2013;<lpage>255</lpage>. <pub-id pub-id-type="doi">10.1007/s11634-013-0158-y</pub-id> </citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Oesterreich</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tseng</surname>
<given-names>G. C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Integrative Clustering of Multi-Level Omics Data for Disease Subtype Discovery Using Sequential Double Regularization</article-title>. <source>Biostat</source> <volume>18</volume>, <fpage>165</fpage>&#x2013;<lpage>179</lpage>. <pub-id pub-id-type="doi">10.1093/biostatistics/kxw039</pub-id> </citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kristensen</surname>
<given-names>V. N.</given-names>
</name>
<name>
<surname>Lingj&#xe6;rde</surname>
<given-names>O. C.</given-names>
</name>
<name>
<surname>Russnes</surname>
<given-names>H. G.</given-names>
</name>
<name>
<surname>Vollan</surname>
<given-names>H. K. M.</given-names>
</name>
<name>
<surname>Frigessi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>B&#xf8;rresen-Dale</surname>
<given-names>A.-L.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Principles and Methods of Integrative Genomic Analyses in Cancer</article-title>. <source>Nat. Rev. Cancer</source> <volume>14</volume>, <fpage>299</fpage>&#x2013;<lpage>313</lpage>. <pub-id pub-id-type="doi">10.1038/nrc3721</pub-id> </citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J. Z.</given-names>
</name>
<name>
<surname>Marron</surname>
<given-names>J. S.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Biclustering via Sparse Singular Value Decomposition</article-title>. <source>Biometrics</source> <volume>66</volume>, <fpage>1087</fpage>&#x2013;<lpage>1095</lpage>. <pub-id pub-id-type="doi">10.1111/j.1541-0420.2010.01392.x</pub-id> </citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lewis</surname>
<given-names>S. J. G.</given-names>
</name>
<name>
<surname>Foltynie</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Blackwell</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Robbins</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Owen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Barker</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Heterogeneity of Parkinson&#x27;s Disease in the Early Clinical Stages Using a Data Driven Approach</article-title>. <source>J. Neurology, Neurosurg. Psychiatry</source> <volume>76</volume>, <fpage>343</fpage>&#x2013;<lpage>348</lpage>. <pub-id pub-id-type="doi">10.1136/jnnp.2003.033530</pub-id> </citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Rowland</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Schrodi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Laird</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tacey</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ross</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2006</year>). <article-title>A Case-Control Association Study of the 12 Single-Nucleotide Polymorphisms Implicated in Parkinson Disease by a Recent Genome Scan</article-title>. <source>Am. J. Hum. Genet.</source> <volume>78</volume>, <fpage>1090</fpage>&#x2013;<lpage>1092</lpage>. <pub-id pub-id-type="doi">10.1086/504725</pub-id> </citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Penalized Feature Selection and Classification in Bioinformatics</article-title>. <source>Briefings Bioinforma.</source> <volume>9</volume>, <fpage>392</fpage>&#x2013;<lpage>403</lpage>. <pub-id pub-id-type="doi">10.1093/bib/bbn027</pub-id> </citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maraganore</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Andrade</surname>
<given-names>M. d.</given-names>
</name>
<name>
<surname>Lesnick</surname>
<given-names>T. G.</given-names>
</name>
<name>
<surname>Pant</surname>
<given-names>P. V. K.</given-names>
</name>
<name>
<surname>Cox</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Ballinger</surname>
<given-names>D. G.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Response from Maraganore et al</article-title>. <source>Am. J. Hum. Genet.</source> <volume>78</volume>, <fpage>1092</fpage>&#x2013;<lpage>1094</lpage>. <pub-id pub-id-type="doi">10.1086/504731</pub-id> </citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maraganore</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>De Andrade</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lesnick</surname>
<given-names>T. G.</given-names>
</name>
<name>
<surname>Strain</surname>
<given-names>K. J.</given-names>
</name>
<name>
<surname>Farrer</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Rocca</surname>
<given-names>W. A.</given-names>
</name>
<etal/>
</person-group> (<year>2005</year>). <article-title>High-resolution Whole-Genome Association Study of Parkinson Disease</article-title>. <source>Am. J. Hum. Genet.</source> <volume>77</volume>, <fpage>685</fpage>&#x2013;<lpage>693</lpage>. <pub-id pub-id-type="doi">10.1086/496902</pub-id> </citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mariette</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Villa-Vialaneix</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Unsupervised Multiple Kernel Learning for Heterogeneous Data Integration</article-title>. <source>Bioinformatics</source> <volume>34</volume>, <fpage>1009</fpage>&#x2013;<lpage>1015</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btx682</pub-id> </citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mortier</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ou&#xe9;draogo</surname>
<given-names>D.-Y.</given-names>
</name>
<name>
<surname>Claeys</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tadesse</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Cornu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Baya</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Mixture of Inhomogeneous Matrix Models for Species-Rich Ecosystems</article-title>. <source>Environmetrics</source> <volume>26</volume>, <fpage>39</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1002/env.2320</pub-id> </citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ndiaye</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Fercoq</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Salmon</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Gap Safe Screening Rules for Sparsity Enforcing Penalties</article-title>. <source>J. Mach. Learn. Res.</source> <volume>18</volume>, <fpage>4671</fpage>&#x2013;<lpage>4703</lpage>. </citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shrestha</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Draghici</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Pinsplus: a Tool for Tumor Subtype Discovery in Integrated Genomic Data</article-title>. <source>Bioinformatics</source> <volume>35</volume>, <fpage>2843</fpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/bty1049</pub-id> </citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rand</surname>
<given-names>W. M.</given-names>
</name>
</person-group> (<year>1971</year>). <article-title>Objective Criteria for the Evaluation of Clustering Methods</article-title>. <source>J. Am. Stat. Assoc.</source> <volume>66</volume>, <fpage>846</fpage>&#x2013;<lpage>850</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.1971.10482356</pub-id> </citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rentzsch</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Witten</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>G. M.</given-names>
</name>
<name>
<surname>Shendure</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kircher</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Cadd: Predicting the Deleteriousness of Variants throughout the Human Genome</article-title>. <source>Nucleic Acids Res.</source> <volume>47</volume>, <fpage>D886</fpage>&#x2013;<lpage>D894</lpage>. <pub-id pub-id-type="doi">10.1093/nar/gky1016</pub-id> </citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ripley</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Venables</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ripley</surname>
<given-names>M. B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Package Nnet</article-title>. <source>R. package</source> <volume>2016</volume>, <fpage>7</fpage>&#x2013;<lpage>3</lpage>. </citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rohart</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gautier</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>L&#xea; Cao</surname>
<given-names>K.-A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>mixOmics: An R Package for &#x27;omics Feature Selection and Multiple Data Integration</article-title>. <source>PLoS Comput. Biol.</source> <volume>13</volume>, <fpage>e1005752</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005752</pub-id> </citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schulam</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Saria</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A Framework for Individualizing Predictions of Disease Trajectories by Exploiting Multi-Resolution Structure</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>2015</volume>, <fpage>748</fpage>&#x2013;<lpage>756</lpage>. </citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mo</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Sparse Integrative Clustering of Multiple Omics Data Sets</article-title>. <source>Ann. Appl. Stat.</source> <volume>7</volume>, <fpage>269</fpage>&#x2013;<lpage>294</lpage>. <pub-id pub-id-type="doi">10.1214/12-AOAS578</pub-id> </citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Olshen</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Ladanyi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Integrative Clustering of Multiple Genomic Data Types Using a Joint Latent Variable Model with Application to Breast and Lung Cancer Subtype Analysis</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>2906</fpage>&#x2013;<lpage>2912</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp543</pub-id> </citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Olshen</surname>
<given-names>A. B.</given-names>
</name>
<name>
<surname>Ladanyi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Integrative Clustering of Multiple Genomic Data Types Using a Joint Latent Variable Model with Application to Breast and Lung Cancer Subtype Analysis</article-title>. <source>Bioinformatics</source> <volume>26</volume>, <fpage>292</fpage>&#x2013;<lpage>293</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btp659</pub-id> </citation>
</ref>
<ref id="B44">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Bi</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Multi-view Sparse Co-clustering via Proximal Alternating Linearized Minimization</article-title>,&#x201d; in <conf-name>Proceedings of the 32nd International Conference on Machine Learning</conf-name>, <conf-loc>Lille, France</conf-loc>, <fpage>757</fpage>&#x2013;<lpage>766</lpage>. <comment>Proceedings of Machine Learning Research, vol. 37</comment>. </citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kranzler</surname>
<given-names>H. R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Multi-view Singular Value Decomposition for Disease Subtyping and Genetic Associations</article-title>. <source>BMC Genet.</source> <volume>15</volume>, <fpage>73</fpage>. <pub-id pub-id-type="doi">10.1186/1471-2156-15-73</pub-id> </citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van der Nest</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lima Passos</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Candel</surname>
<given-names>M. J. J. M.</given-names>
</name>
<name>
<surname>van Breukelen</surname>
<given-names>G. J. P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An Overview of Mixture Modelling for Latent Evolutions in Longitudinal Data: Modelling Approaches, Fit Statistics and Software</article-title>. <source>Adv. Life Course Res.</source> <volume>43</volume>, <fpage>100323</fpage>. <pub-id pub-id-type="doi">10.1016/j.alcr.2019.100323</pub-id> </citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yi</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Caramanis</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Regularized Em Algorithms: A Unified Framework and Statistical Guarantees</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>28</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. </citation>
</ref>
<ref id="B48">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kwok</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Multiple Kernel Clustering</article-title>,&#x201d; in <conf-name>Proceedings of the SIAM International Conference on Data Mining</conf-name>, <fpage>638</fpage>&#x2013;<lpage>649</lpage>. <pub-id pub-id-type="doi">10.1137/1.9781611972795.55</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>