<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2024.1476070</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Disentangling genotype and environment specific latent features for improved trait prediction using a compositional autoencoder</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Powadi</surname>
<given-names>Anirudha</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2810742"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jubery</surname>
<given-names>Talukder Zaki</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/381267"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Tross</surname>
<given-names>Michael C.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2666319"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Schnable</surname>
<given-names>James C.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/22976"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ganapathysubramanian</surname>
<given-names>Baskar</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/483726"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Electrical and Computer Engineering, Iowa State University</institution>, <addr-line>Ames, IA</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Translational AI Research and Education Center, Iowa State University</institution>, <addr-line>Ames, IA</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Agronomy and Horticulture, University of Nebraska-Lincoln</institution>, <addr-line>Lincoln, NE</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Center for Plant Science Innovation, University of Nebraska-Lincoln</institution>, <addr-line>Lincoln, NE</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Mechanical Engineering, Iowa State University</institution>, <addr-line>Ames, IA</addr-line>, <country>United States</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Plant Science Institute, Iowa State University</institution>, <addr-line>Ames, IA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Andr&#xe9;s J. Cort&#xe9;s, Colombian Corporation for Agricultural Research (AGROSAVIA), Colombia</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Sherif El-Areed, Beni-Suef University, Egypt</p>
<p>Joaquin Guillermo Ramirez Gil, National University of Colombia, Colombia</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: James C. Schnable, <email xlink:href="mailto:schnable@unl.edu">schnable@unl.edu</email>; Baskar Ganapathysubramanian, <email xlink:href="mailto:baskarg@iastate.edu">baskarg@iastate.edu</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>12</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1476070</elocation-id>
<history>
<date date-type="received">
<day>05</day>
<month>08</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>11</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Powadi, Jubery, Tross, Schnable and Ganapathysubramanian</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Powadi, Jubery, Tross, Schnable and Ganapathysubramanian</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>In plant breeding and genetics, predictive models traditionally rely on compact representations of high-dimensional data, often using methods like Principal Component Analysis (PCA) and, more recently, Autoencoders (AE). However, these methods do not separate genotype-specific and environment-specific features, limiting their ability to accurately predict traits influenced by both genetic and environmental factors. We hypothesize that disentangling these representations into genotype-specific and environment-specific components can enhance predictive models. To test this, we developed a compositional autoencoder (CAE) that decomposes high-dimensional data into distinct genotype-specific and environment-specific latent features. Our CAE framework employed a hierarchical architecture within an autoencoder to effectively separate these entangled latent features. Applied to a maize diversity panel dataset, the CAE demonstrated superior modeling of environmental influences and out-performs PCA (principal component analysis), PLSR (Partial Least square regression) and vanilla autoencoders by 7 times for &#x2018;Days to Pollen&#x2019; trait and 10 times improved predictive performance for &#x2018;Yield&#x2019;. By disentangling latent features, the CAE provided a powerful tool for precision breeding and genetic research. This work has significantly enhanced trait prediction models, advancing agricultural and biological sciences.</p>
</abstract>
<kwd-group>
<kwd>hierarchical disentanglement</kwd>
<kwd>latent disentanglement</kwd>
<kwd>plant phenotyping</kwd>
<kwd>days to pollen</kwd>
<kwd>yield</kwd>
<kwd>GxE</kwd>
</kwd-group>
<contract-num rid="cn001">2021-67021-35329</contract-num>
<contract-sponsor id="cn001">National Institute of Food and Agriculture<named-content content-type="fundref-id">10.13039/100005825</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Plant Sciences Institute, Iowa State University<named-content content-type="fundref-id">10.13039/100015727</named-content>
</contract-sponsor>
<counts>
<fig-count count="9"/>
<table-count count="12"/>
<equation-count count="4"/>
<ref-count count="51"/>
<page-count count="13"/>
<word-count count="6146"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Plant Breeding</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Advances in imaging and robotic technologies are making both high-resolution images and sensor data increasingly accessible to plant biologists and breeders as tools to capture measurements of plant traits. These data types can be used to measure or predict traits that are labor-intensive or costly to measure directly, including variation in plant architectural and biochemical traits as well as resistance or susceptibility to specific biotic stresses. A growing body of evidence suggests high dimensional trait datasets can also be useful to predict crop productivity (e.g. grain yield) (<xref ref-type="bibr" rid="B1">Adak et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B21">Jin et&#xa0;al., 2024</xref>). However, like the plant traits plant biologists and breeders seek to predict, sensor data and the high dimensional traits extracted from that data reflect the impact of both genetic and environmental factors.</p>
<p>Traditionally, such data are analyzed in raw form or by using handcrafted features without explicitly separating genotype (G) and environment (E) factors. Handcrafting features for high-dimensional data can be challenging due to the &#x2018;curse of dimensionality,&#x2019; where increasing complexity hinders interpretability, accuracy, and generalizability of models across environments and genotypes. In contrast, latent features derived from unsupervised learning methods capture underlying patterns without the biases of human assumptions, providing more generalizable models for predicting complex traits (<xref ref-type="bibr" rid="B14">Feldmann et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B2">Aguate et&#xa0;al., 2017</xref>).</p>
<p>Latent phenotyping has emerged as a promising approach to minimize human bias by reducing data dimensionality via unsupervised or self-supervised approaches (<xref ref-type="bibr" rid="B15">Gage et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B41">Ubbens et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B14">Feldmann et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B40">Tross et&#xa0;al., 2023</xref>). Traditionally, machine learning methods like PCA (Principal component analysis), Linear Discriminant Analysis (LDA), T-distributed Stochastic Neighbor Embedding (t-SNE), and autoencoders have been used to extract the &#x2018;latent representation&#x2019; from high-dimensional data (<xref ref-type="bibr" rid="B3">Alexander et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B51">Zhong et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B28">Kopf and Claassen, 2021</xref>; <xref ref-type="bibr" rid="B36">Song et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B16">Gomari et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B19">Iwasaki et&#xa0;al., 2023</xref>). Autoencoders, in particular, offer advantages in capturing non-linear relationships. By compressing data into a latent space and reconstructing the original input, autoencoders learn a compact yet informative representation crucial for phenotyping (<xref ref-type="bibr" rid="B15">Gage et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B41">Ubbens et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B40">Tross et&#xa0;al., 2023</xref>). Autoencoder-derived representations, though informative, often fail to separate genotype and environment influences, leading to &#x2018;entangled&#x2019; latent spaces where distinct plant attributes, such as &#x2018;leaf number,&#x2019; &#x2018;height,&#x2019; and &#x2018;chlorophyll concentration,&#x2019; are intermixed rather than independently represented. Disentangling these attributes within the latent space can improve latent factors&#x2019; interpretability.</p>
<p>Our hypothesis is that disentangling genotype and environment effects within the latent space can improve prediction accuracy and enhance model generalizability to new genotypes and environments. Specifically, we aim to separate environmental factors (e.g., soil conditions, weather, treatment) and genetic influences in high-dimensional hyperspectral data representing maize phenotypes. We believe that disentangling the latent space into environment and gene effects should help improve the predictive performance of the learned representation on many downstream tasks, as shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Trait prediction workflow of a Vanilla Autoencoder vs Compositional Autonencoder.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g001.tif"/>
</fig>
<p>Several disentanglement methods have been proposed, though they often compromise reconstruction accuracy. A common strategy involves regularization techniques, where additional terms in the loss function, as seen in variational autoencoders (VAEs) (<xref ref-type="bibr" rid="B26">Kingma and Welling, 2019</xref>), encourage independence among latent variables. For example, <italic>&#x3b2;</italic>-VAE (<xref ref-type="bibr" rid="B18">Higgins et&#xa0;al., 2017</xref>) balances reconstruction and disentanglement, while FactorVAE (<xref ref-type="bibr" rid="B24">Kim and Mnih, 2019</xref>) uses total correlation penalties to promote variable independence. Mutual information-based approaches, such as InfoGAN and StyleGAN, enhance disentanglement by maximizing the distinctiveness of latent factors in the output, and supervised or semi-supervised techniques leverage labeled data to guide disentangled representation learning (<xref ref-type="bibr" rid="B29">Kulkarni et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B25">Kingma et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B27">Kingma and Welling, 2022</xref>).</p>
<p>Disentanglement approaches fall broadly into hierarchical and latent space methods. Hierarchical disentanglement organizes the latent space into levels, where higher layers capture abstract features and lower layers focus on specific details. Latent space disentanglement, in contrast, promotes independent variation by assigning each latent dimension to a distinct feature (<xref ref-type="bibr" rid="B8">Burgess et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B50">Zheng and Sun, 2019</xref>; <xref ref-type="bibr" rid="B43">Watters et al., 2019</xref>; <xref ref-type="bibr" rid="B9">Cha and Thiyagalingam, 2023</xref>). StyleGAN (<xref ref-type="bibr" rid="B31">Liu et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B34">Niu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B45">Wei et&#xa0;al., 2023</xref>) achieves this by associating unique features with specific components of a Gaussian latent vector, while hierarchical disentanglement has been applied across domains, including speech (<xref ref-type="bibr" rid="B38">Sun et&#xa0;al., 2020</xref>), video sequences (<xref ref-type="bibr" rid="B11">Comas et&#xa0;al., 2021</xref>), and multi-modal data (<xref ref-type="bibr" rid="B10">Chen and Zhang, 2023</xref>) using attention (<xref ref-type="bibr" rid="B12">Cui et&#xa0;al., 2024</xref>), context addition (<xref ref-type="bibr" rid="B30">Li et&#xa0;al., 2021</xref>), graph convolution (<xref ref-type="bibr" rid="B5">Bai et&#xa0;al., 2022</xref>), and contrastive learning (<xref ref-type="bibr" rid="B46">Xie et&#xa0;al., 2023</xref>).</p>
<p>Orthogonal denoising autoencoders (<xref ref-type="bibr" rid="B47">Ye et&#xa0;al., 2016</xref>) and factorized latent space models (<xref ref-type="bibr" rid="B20">Jia et&#xa0;al., 2010</xref>) enhance disentanglement by learning features from multiple perspectives within a dataset, enabling the integration of diverse data sources. Additionally, correlation loss has been applied to effectively separate identity and expression in facial representations (<xref ref-type="bibr" rid="B37">Sun et&#xa0;al., 2019</xref>). Latent feature disentanglement has found applications across various fields, including music (<xref ref-type="bibr" rid="B7">Banar et&#xa0;al., 2023</xref>), text (<xref ref-type="bibr" rid="B42">Wang et&#xa0;al., 2022</xref>), facial generation (<xref ref-type="bibr" rid="B23">Karras et&#xa0;al., 2019</xref>), and protein structure variation (<xref ref-type="bibr" rid="B39">Tatro et&#xa0;al., 2021</xref>), though its use in plant phenotyping remains limited.</p>
<p>In this paper, we propose a compositional autoencoder (CAE), inspired by orthogonal denoising autoencoders (<xref ref-type="bibr" rid="B47">Ye et&#xa0;al., 2016</xref>) and factorized latent space models (<xref ref-type="bibr" rid="B20">Jia et&#xa0;al., 2010</xref>), to disentangle genotype and environment effects within the latent space. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> illustrates the problem definition of the disentangled latent space representation, where environmental factors can include a range of variables such as weather, soil conditions, and treatments applied to plants in a field. Our objectives in this work are as follows:</p>
<list list-type="bullet">
<list-item>
<p>Develop a compositional autoencoder (CAE) to separate genotype-specific, macro-, and microenvironmental effects in hyperspectral data.</p>
</list-item>
<list-item>
<p>Assess whether CAE-generated latent representations improve predictive accuracy for traits like Days to Pollen and Yield.</p>
</list-item>
<list-item>
<p>Examine the consistency of the CAE&#x2019;s performance across different model initializations and hyperparameters for potential applications in trait prediction.</p>
</list-item>
</list>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Problem definition: Disentangling genotype-specific, environment-specific, and plant-specific information from hyperspectral data. The goal is to separate features associated with genotype, field-level environmental conditions, and individual plant variations across multiple environments and replicates. This achieved by the method of composition.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g002.tif"/>
</fig>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Equipment and dataset</title>
<p>Hyperspectral data is being increasingly adopted by plant scientists as a method to measure or predict plant traits in field and greenhouse settings (<xref ref-type="bibr" rid="B22">Kaleita et&#xa0;al., 2006</xref>; <xref ref-type="bibr" rid="B49">Zhang et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B48">Yendrek et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B40">Tross et&#xa0;al., 2023</xref>). For the purposes of this study, we employed data from 578 inbreds, which represent a subset of the Wisconsin Diversity panel (<xref ref-type="bibr" rid="B32">Mazaheri et&#xa0;al., 2019</xref>), grown and phenotyped in 2020 and 2021 at the Havelock Farm research facility at the University of Nebraska-Lincoln. In each year, measurements were collected on two replicated plots of each inbred grown in different parts of the field, for a total 2&#xd7;2&#xd7;578 = 2312 observed plots. Each plot consisted of two rows of genetically identical plants with approximately 20 plants per row, as previously described in <xref ref-type="bibr" rid="B33">Mural et&#xa0;al. (2022)</xref>. Hyperspectral data was collected using FieldSpec4 spectroradiometers (Malvern Panalytical Ltd., Formerly Analytical Spectral Devices) with a contact probe. This equipment captures 2151 wavelengths of electromagnetic radiation ranging from 350 nm to 2500 nm. Hyperspectral data was collected from a single fully expanded leaf per plot, selected from a representative plant, avoiding edge plants whenever possible. Three spectral measurements were taken at each of the three points located at the tip, middle, and base of the adaxial side of each leaf. Values were averaged across the nine wavelength scans to generate a final composite spectrum for each plot sampled (<xref ref-type="bibr" rid="B40">Tross et&#xa0;al., 2023</xref>). <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> illustrates the distribution and variability of mean reflectance among the genotypes across two years, which in this paper are referred to as two different environments. We divide the environment into field-level (or macro-environment) and plot-level (or micro-environment) <xref ref-type="bibr" rid="B17">Guil et&#xa0;al. (2009)</xref>. For the latent features extraction, the data was then normalized using min-max normalization. This normalization is given as:</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Hyperspectral leaf reflectance data was collected using a FieldSpec4 (Malvern Panalytical Ltd., Formerly Analytical Spectral Devices) with a contact probe. A total of 2151 wavelengths were collected, ranging from 350 nm to 2500 nm. The dataset consists of measurements for a set of 578 different maize inbred genotypes that were grown and phenotyped in two different environments with 2 replicates per environment.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g003.tif"/>
</fig>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mtext>normalized</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mtext>dataset</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mtext>dataset</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mtext>dataset</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>From the <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>, &#x2018;<inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>&#x2019; and &#x2018;<inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>&#x2019; are the minimum and maximum values in the entire dataset respectively.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Vanilla autoencoder</title>
<p>We implemented a standard autoencoder (see <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>) as a baseline for comparison which we refer to below as the &#x2018;vanilla autoencoder&#x2019; (AE). Both the encoder and decoder portions of our vanilla autoencoder implementation are made up of multiple fully connected layers stacked together with the non-linear activation function &#x2018;SeLu.&#x2019; The encoder encodes the input data (2151 wavelengths) into smaller dimensions (latent space) and decoder works to reconstruct back the original input from this latent space. The <xref ref-type="table" rid="T1">
<bold>Tables&#xa0;1</bold>
</xref>, <xref ref-type="table" rid="T2">
<bold>2</bold>
</xref> show the details of each of the layers that constitute the encoder and decoder. For training the vanilla autoencoder, data from each plot in each year is considered as one sample, resulting in a total of 2312 input samples.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>A vanilla autoencoder works to learn a compressed yet highly informative representation of the input data.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g004.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Encoder: Configuration details.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Layer Type</th>
<th valign="top" align="center">Dimensions</th>
<th valign="top" align="center">Activation</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">input_shape &#x2192; 2150</td>
<td valign="top" align="center">SELU</td>
</tr>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">2150 &#x2192; 1024</td>
<td valign="top" align="center">SELU</td>
</tr>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">1024 &#x2192; 512</td>
<td valign="top" align="center">SELU</td>
</tr>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">512 &#x2192; zg + ze + zp</td>
<td valign="top" align="center">None</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x2018;input_shape&#x2019; = 1 x 2151, &#x2018;zg&#x2019; = dimensions allocated to capture genotype features, &#x2018;ze&#x2019; = dimensions allocated to capture macro-environment features, &#x2018;zp&#x2019; = dimensions allocated to micro-environment features.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Decoder: Configuration details.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Layer Type</th>
<th valign="top" align="center">Dimensions</th>
<th valign="top" align="center">Activation</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">zg + ze + zp &#x2192; 512</td>
<td valign="top" align="center">SELU</td>
</tr>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">512 &#x2192; 1024</td>
<td valign="top" align="center">SELU</td>
</tr>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">1024 &#x2192; 2150</td>
<td valign="top" align="center">SELU</td>
</tr>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">2150 &#x2192; input_shape</td>
<td valign="top" align="center">Sigmoid</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x2018;input_shape&#x2019; = 1 x 2151, &#x2018;zg&#x2019; = dimensions allocated to capture genotype features, &#x2018;ze&#x2019; = dimensions allocated to capture macro-environment features, &#x2018;zp&#x2019; = dimensions allocated to micro-environment features.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Compositional autoencoder</title>
<sec id="s2_3_1">
<label>2.3.1</label>
<title>Architecture</title>
<p>The compositional autoencoder extends the vanilla autoencoder architecture in a way that aims to disentangle the latent space, partitioning the impact of different factors that influence the data into different variables. It consists of an encoder, decoder, and a fusion block. The network operates as follows:</p>
<list list-type="order">
<list-item>
<p>Encode Individual Plant Data: The encoder processes data from four plants of the same genotype, compressing it into latent features.</p>
</list-item>
<list-item>
<p>Fuse Encoded Data: These encoded representations from all the plants are then fused into a single latent feature.</p>
</list-item>
<list-item>
<p>Disentangle Latent Factors: This fused latent feature is then partitioned into three distinct parts: genotype-specific features (common across all plants), macro-environment-specific features (shared by plants from the same environment), and micro-environment-specific features (unique to each plant).</p>
</list-item>
<list-item>
<p>Reconstruct Individual Plants: Finally, for each plant, the genotype, macro-environment, and micro-environment features are assembled. This assembled disentangled representation is then decoded to reconstruct the original plant data.</p>
</list-item>
</list>
<p>Here, genotype refers to groups of plants with identical genetic makeups, macro-environment refers to common environmental factors experienced by all plants growing in the same field in the same year (e.g. rainfall, temperature), and micro-environment refers to features of the individual replicate growing in the same field within the same environment/year. The table (refer to <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>) illustrates the disentangled latent representation for each plant. A more detailed network architecture can be found in the figure (refer to <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>). The encoder and decoder used here are the same as vanilla autoencoder with the addition of &#x2018;Fusion&#x2019; layer. The layer details are provided in the <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Disentangled latent-space representation of each plant.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Plant</th>
<th valign="top" align="center">Representation</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Plant 1</td>
<td valign="top" align="center">{(Zg) genotype, (Ze) macro-environment [1], (Zp) micro-environment [1]}</td>
</tr>
<tr>
<td valign="top" align="center">Plant 2</td>
<td valign="top" align="center">{(Zg) genotype, (Ze) macro-environment [1], (Zp) micro-environment [2]}</td>
</tr>
<tr>
<td valign="top" align="center">Plant 3</td>
<td valign="top" align="center">{(Zg) genotype, (Ze) macro-environment [2], (Zp) micro-environment [3]}</td>
</tr>
<tr>
<td valign="top" align="center">Plant 4</td>
<td valign="top" align="center">{(Zg) genotype, (Ze) macro-environment [2], (Zp) micro-environment [4]}</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>The encoder encodes the hyperspectral data for 4 plants, accounting for a single genotype across two environments (E1, and E2) and 2 replicated per environment (P1E1, P2E1, P1E2, P2E2). The resulting 4 latent vectors are fused using a linear layer. The resulting fused vector contains 3 parts. (1) Genotype representation part. (2) Macro or field-level environment representation part (2 parts to represent E1 and E2 effects). (3) Micro or replicate-specific environment representation component [4 parts to represent each of the plants (P1E1, P2E1, P1E2, P2E2)]. To get the composed encoded form, genotype representation is combined with the field-level environment part and plant-level environment part. These composed encoded vectors are then fed into the decoder to regenerate the original hyperspectral reflectance.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g005.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Fusion layer details.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Layer Type</th>
<th valign="top" align="center">Dimensions</th>
<th valign="top" align="center">Activation</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Linear</td>
<td valign="top" align="center">N(zg + ze + zp) &#x2192; zg + E(ze) + N(zp)</td>
<td valign="top" align="center">None</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>&#x2018;N&#x2019; = number of replicates per genotype (2), &#x2018;E&#x2019; = number of environments. (2), &#x2018;zg&#x2019; = dimensions allocated to capture genotype features, &#x2018;ze&#x2019; = dimensions allocated to capture macro-environment features, &#x2018;zp&#x2019; = dimensions allocated to micro-environment features.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The training process involves dividing the hyperspectral data into groups of four plants (sharing the same genotype). There are a total of 578 such groups (corresponding to the number of genotypes). Each group is fed sequentially through the encoder, resulting in four latent representations. These representations are then fused together. The resulting fused latent space captures three factors: genotype, field-level environment (with two sub-parts for the two environments), and plant-level environment (with four sub-parts for the four plants).</p>
</sec>
<sec id="s2_3_2">
<label>2.3.2</label>
<title>Loss function</title>
<p>We trained the CAE network using a two-part loss function consisting of a reconstruction loss and a correlation loss.</p>
<p>
<italic>Reconstruction Loss:</italic> The mean squared error (MSE), was used as the reconstruction loss for the compositional autoencoder. This loss function encourages the network to learn a meaningful disentangled latent space that can be accurately decoded back to the original hyperspectral data.</p>
<p>
<italic>Correlation Loss:</italic> A correlation loss was employed to ensure that all parts in the disentangled latent space remain uncorrelated throughout the training process. This loss is defined in <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>.</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtext>Correlation&#xa0;Loss</mml:mtext>
<mml:mo>=</mml:mo>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CorrMat</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where:</p>
<list list-type="bullet">
<list-item>
<p>
<inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>CorrMat</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the correlation coefficient between dimensions <italic>i</italic> and <italic>j</italic> in the latent space.</p>
</list-item>
<list-item>
<p>
<italic>N</italic> is the dimension of the square correlation matrix, which corresponds to the number of dimensions in the latent space.</p>
</list-item>
<list-item>
<p>
<inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the identity matrix, ensuring that the diagonal elements (where <italic>i</italic> = <italic>j</italic>) contribute zero to the loss.</p>
</list-item>
</list>
<p>The correlation coefficient used here is the Pearson correlation coefficient (<italic>r</italic>), a measure of the linear correlation between two variables. It is calculated using <xref ref-type="disp-formula" rid="eq3">Equation 3</xref>.</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>p</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>k</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>p</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>k</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where:</p>
<list list-type="bullet">
<list-item>
<p>
<italic>n</italic> is the number of data points.</p>
</list-item>
<list-item>
<p>
<inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>k</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the elements of the latent space.</p>
</list-item>
<list-item>
<p>
<inline-formula>
<mml:math display="inline" id="im7">
<mml:mover accent="true">
<mml:mi>p</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im8">
<mml:mover accent="true">
<mml:mi>k</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> are the means of the <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> dimension and <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> dimension, respectively.</p>
</list-item>
</list>
<p>In our case, we aim to achieve zero correlation between the latent space features representing genotype, environment, and individual plant variations. This is enforced by the correlation loss function (<xref ref-type="disp-formula" rid="eq2">Equation 2</xref>). This ensures that the disentangled latent space captures these factors independently.</p>
<p>We trained the vanilla autoencoder network using MSE reconstruction loss only.</p>
</sec>
<sec id="s2_3_3">
<label>2.3.3</label>
<title>Training parameters</title>
<p>The data was divided into training and validation with a 85%-15% split. Furthermore, we trained these networks with SGD, Adam, and LBFGS optimizers and found that LBFGS gave us faster convergence (10x). Therefore, all the experiments were carried out using the LBFGS optimizer. The training setup included early stopping criteria, which monitored validation loss and stopped training after it observed no improvements in the metric for 15 epochs.</p>
</sec>
<sec id="s2_3_4">
<label>2.3.4</label>
<title>Parameter tuning for downstream tasks</title>
<p>To improve the performance of latent representations for downstream tasks, we investigated several tuning techniques for both the network and its inputs.</p>
<list list-type="bullet">
<list-item>
<p>a) We explored masking a portion of the input data. This technique encourages the model to focus on reconstructing the missing parts, potentially leading to increased robustness and reduced overfitting (<xref ref-type="bibr" rid="B4">Bachmann et&#xa0;al., 2022</xref>). We performed a search for the optimal masking percentage.</p>
</list-item>
<list-item>
<p>b) Considering our dataset size, we conducted a basic architecture search to strike a balance between model complexity and data availability. This helps to mitigate overfitting and improve generalization. We evaluated different network architectures with varying numbers of layers and dimensions in the encoder and decoder.</p>
</list-item>
<list-item>
<p>c) To ensure the latent representations captured the necessary data complexity, we experimented with different latent space dimensions and their composition of genotype, field-level, and plant-level environmental features.</p>
</list-item>
</list>
</sec>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Downstream tasks performance metrics</title>
<p>To confirm our hypothesis that the disentangled latent representations enhance the latent feature&#x2019;s ability to predict useful traits, we generated disentangled latent features (disentangled encoded output from the encoder) for all 2312 data points. We then used these features to train models to predict two traits, namely, &#x2018;Days to Pollen&#x2019; and &#x2018;Yield (grams)&#x2019;. We trained several regression models &#x2014; Random Forests, XGBoost, Ridge Regressions, and PLSR (Partial-Least Square Regression) &#x2014; to identify a high performing model. We compare the performance of the models trained on the disentangled latent representations from the CAE against the performance of models trained on the latent representations from a vanilla autoencder. The resulting prediction performance was evaluated using an R<sup>2</sup> metric representing the coefficient of determination. The coefficient of determination, <italic>R</italic>
<sup>2</sup>, is defined as:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where:</p>
<list list-type="bullet">
<list-item>
<p>
<inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the observed value,</p>
</list-item>
<list-item>
<p>
<inline-formula>
<mml:math display="inline" id="im13">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> is the predicted value, and</p>
</list-item>
<list-item>
<p>
<inline-formula>
<mml:math display="inline" id="im14">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> is the mean of the observed data.</p>
</list-item>
</list>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and discussion</title>
<sec id="s3_1">
<label>3.1</label>
<title>Disentangled representation from CAE</title>
<p>The compositional autoencoder (CAE) successfully disentangled the latent space into genotype, macro-and micro- environmental effects. The <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> shows a comparison of the original reflectance versus factor-specific (genotype and environments) reflectance. Here, factor-specific reflectance is obtained by modifying the latent space to only keep the effects of either the genotype, or the environments; and subsequently reconstructing the reflectance from them. Therefore, genotype-specific is obtained by replacing the environment components in the latent space with an average of all the environments, and similarly, genotype components are replaced by their average to reconstruct the environment-specific reflectance. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref> shows genotype-specific reflectance. As we are focusing on just 1 genotype in this figure, all the replicates will have the same latent space and therefore, the same reflectance. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6C</bold>
</xref> shows macro environment-specific reflectance. The distinction between the two macro-environments is visualized by calculating the difference between macro-environment-specific reflectance and genotype-specific reflectance for the two macro-environments. Similarly, <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6D</bold>
</xref> shows micro-environment-specific reflectance. The visualization shows the difference between genotype-specific reflectance, macro-environment-specific reflectance, and micro-environment-specific reflectance.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Reflectance Measurements and Disentangled Influences: <bold>(A)</bold> the original measured reflectance spectra for multiple samples of a particular genotype, and the disentangled reflectance components attributed to <bold>(B)</bold> genotype, <bold>(C)</bold> macro-environmental influence, and <bold>(D)</bold> micro-environmental influence. Note the significant variation in the original reflectance due to the combined effects of genotype and environment. Disentanglement enables the visualization of distinct spectral patterns associated with each factor, highlighting the CAE&#x2019;s ability to separate these influences.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g006.tif"/>
</fig>
<p>To further verify the degree of environment disentanglement, we calculated the distribution of the two macro environments for the original reflectance (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7A</bold>
</xref>) and disentangled environments&#x2019; reflectance (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7B</bold>
</xref>). A successful disentanglement should yield completely separated distributions. We use KL-divergence to measure the difference between the distributions. We can clearly see that KL-divergence of distributions representing two environments generated from the sensor data is quite low (0.62) while the same for the disentangled reflectance is quite large (2.79). This strongly indicates that the latent representation is, in fact, able to represent the two environments distinctly.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Two sets of visualizations are presented: <bold>(A)</bold> the original environmental effects (Env 1 and Env 2), and <bold>(B)</bold> the disentangled versions. The average KL-divergence observed for the original input data is 0.62, while the disentangled KL-divergence is 2.79. The density distribution of reflectance values is shown at selected wavelengths for clarity, illustrating the separation of environmental factors before and after applying the CAE.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g007.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Performance of latent representations on downstream tasks</title>
<p>We first report on the performance of our baseline model &#x2013; the vanilla autoencoder. The latent representation from the vanilla AE was used to train a multiple machine learning models to predict the two traits. We present the Ridge regression model performance here as it yielded the best results among all the models (Random Forests, PLSR, and XgBoost). <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> shows this performance. We see that the performance for both the traits in question is quite low (<italic>r</italic>
<sup>2</sup> = 0.01).</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Performance of AE on &#x2018;Days to Pollen&#x2019; and &#x2018;Yield&#x2019; using Ridge regression: <bold>(A)</bold> Days to Pollen: The x-axis represents ground truth days (60-90 days), and the y-axis represents predicted days (55-90 days). The scatter plot shows points widely scattered, indicating poor prediction accuracy. The model achieves an R of -0.01 (0.01), demonstrating negligible correlation between predicted and actual values. <bold>(B)</bold> Total Grain Mass: The x-axis represents ground truth grain mass (200-1000 grams), and the y-axis represents predicted grain mass (200-1000 grams). The scatter plot shows points widely scattered, indicating poor prediction accuracy. The model achieves an R of -0.01 (0.02), demonstrating negligible correlation between predicted and actual values.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g008.tif"/>
</fig>
<p>Next, we compare this against the performance of the CAE based disentangled representation (similarly trained with multiple machine learning models out of which XgBoost yielded the best results and its performance is reported here). <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> shows the performance of the structured latent representation generated by the CAE. The Compositional Autoencoder (CAE) performs exceptionally well for the &#x2018;Days to Pollen&#x2019; trait, achieving an <italic>r</italic>
<sup>2</sup> value of 0.74. While its performance in predicting &#x2018;Yield&#x2019; is lower, with an <italic>r</italic>
<sup>2</sup> value of 0.34, this is unsurprising given the complexity of the genetic architecture governing yield. Accurate prediction of yield is inherently challenging due to its intricate genetic influences. Previous studies with these genotypes (<xref ref-type="bibr" rid="B21">Jin et&#xa0;al., 2024</xref>) involved costly and labor-intensive genotyping and manual trait measurements. These methods require significant time and effort. Considering these factors, achieving such performance using leaf hyperspectral reflectance collected only at a single time point is significant.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Performance of CAE on &#x2018;Days to Pollen&#x2019; and &#x2018;Yield&#x2019; using Xg-Boost: <bold>(A)</bold> Days to Pollen: The x-axis indicates ground truth days (60-90 days), and the y-axis indicates predicted days (55-90 days). Points near the line y = x indicate accurate predictions. The model achieves an R of 0.74 (0.03), demonstrating strong prediction accuracy. <bold>(B)</bold> Yield: The x-axis indicates ground truth yield in grain mass (200-1000 grams), and the y-axis indicates predicted grain mass (200-1000 grams). Points near the line y = x indicate accurate predictions. The model achieves an R of 0.34 (0.06), demonstrating moderate prediction accuracy.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1476070-g009.tif"/>
</fig>
<p>It is worthwhile to compare these results against recent studies based on collecting hyperspectral reflectance measurements of whole canopies instead of the leaf reflectance used here. However, we were unable to find studies reporting results on a diversity panel, so direct comparison is very difficult. The closest was work by <xref ref-type="bibr" rid="B13">Fan et&#xa0;al. (2022)</xref>, who reported a <italic>r</italic>
<sup>2</sup> = 0.29 and <italic>r</italic>
<sup>2</sup> = 0.84 for predicting &#x2018;yield&#x2019; and &#x2018;Days to Pollen&#x2019;, respectively, from hyperspectral imagery of the Genomes2Field project, which consists of around 1000 hybrids. <xref ref-type="bibr" rid="B6">Baio et&#xa0;al. (2023)</xref> used hyperspectral images of the canopy of a single commercial hybrid across multiple environments to predict yield with <italic>r</italic>
<sup>2</sup> = 0.33 with a random forest model. We see that using the CAE approach on leaf scale phenotyping produces competitive results compared to state-of-the-art canopy scale phenotyping. Recent work also suggests that using the hyperspectral data to infer intermediate physiological parameters that are subsequently used to predict yield is a promising approach. For instance, <xref ref-type="bibr" rid="B44">Weber et&#xa0;al. (2012)</xref>, used leaf reflectance and canopy reflectance to get an <italic>r</italic>
<sup>2</sup> = 0.7 for leaf reflectance of 100 genotypes. Our findings suggest that CAE-generated latent representations hold promise for capturing relevant yield-related information. Further research is needed to explore the integration of these latent representations with other data sources to potentially improve yield prediction accuracy.</p>
<p>Finally, we compared the effectiveness of using latent representations from (a) a Principal Component Analysis (PCA) on raw data, (b) latent representations from a vanilla autoencoder (AE), and (c) latent representations from a compositional autoencoder (CAE) for predicting the traits of &#x2018;Days to Pollen&#x2019; and &#x2018;Yield&#x2019;. Here, we aim to assess whether the learned latent representations offer benefits compared to using the original data directly.</p>
<p>
<xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> (yield) and 6 (days to pollen) summarize the performance comparison using the R-squared metric (coefficient of determination) using a 5-fold cross-validation process. The tables showcase the average R-squared values (with standard deviation in parenthesis) achieved by each method and the best-performing machine learning model for that particular scenario. The performances of all the models has been given in the <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref> section.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>A final comparison between baseline (PCA on raw data), vanilla autoencoder, and compositional autoencoder for yield prediction.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Metric - Model</th>
<th valign="top" align="center">Avg. Values</th>
<th valign="top" align="center">ML Model</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> &#x2212; <italic>CAE</italic>
</td>
<td valign="top" align="center">0.351 (0.058)</td>
<td valign="top" align="center">Xg-Boost Regression</td>
</tr>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> &#x2212; <italic>AE</italic>
</td>
<td valign="top" align="center">0.026 (0.017)</td>
<td valign="top" align="center">Ridge Regression</td>
</tr>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> &#x2212; <italic>PCA</italic>
</td>
<td valign="top" align="center">0.034 (0.016)</td>
<td valign="top" align="center">Ridge Regression</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As observed in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>, the CAE achieves a significantly higher average R-squared value (0.351) compared to both the AE (0.026) and the baseline using PCA on raw data (0.034) for predicting &#x201c;Yield.&#x201d; This suggests that the disentangled latent representations learned by the CAE capture more relevant information for predicting yield compared to the other methods. The best performing model for all three scenarios is Xg-Boost Regression, highlighting its effectiveness for this particular regression task.</p>
<p>Similarly, <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref> shows the results for predicting &#x201c;Days to Pollen.&#x201d; Here, CAE again demonstrates a clear advantage with an average R-squared value of 0.68, significantly higher than both AE (0.106) and the baseline PCA approach (0.108). This reinforces the notion that the disentangled representations from the CAE do a better job of capturing the factors influencing the number of days to pollen in the data.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>A final comparison between baseline (PCA on raw data), vanilla autoencoder, and compositional autoencoder for Days to Pollen.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Metric - Model</th>
<th valign="top" align="center">Avg. Values</th>
<th valign="top" align="center">ML Model</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> &#x2212; <italic>CAE</italic>
</td>
<td valign="top" align="center">0.68 (0.034)</td>
<td valign="top" align="center">Xg-Boost Regression</td>
</tr>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> &#x2212; <italic>AE</italic>
</td>
<td valign="top" align="center">-0.01 (0.025)</td>
<td valign="top" align="center">Ridge Regression</td>
</tr>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> &#x2212; <italic>RAW</italic> &#x2212; <italic>PCA</italic>
</td>
<td valign="top" align="center">0.108 (0.02)</td>
<td valign="top" align="center">Ridge Regression</td>
</tr>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> &#x2212; <italic>RAW</italic>
</td>
<td valign="top" align="center">0.16 (0.00)</td>
<td valign="top" align="center">Ridge Regression</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Overall, these results suggest that leveraging the latent representations learned by the CAE offers a substantial advantage for predicting both &#x201c;Yield&#x201d; and &#x201c;Days to Pollen&#x201d; compared to using the raw data directly or latent representations from the AE. This highlights the effectiveness of disentangled representations in capturing underlying factors that are relevant to these specific traits.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Consistency of latent representations</title>
<p>We evaluate the consistency of the disentangled latent representations by training the model with multiple initial conditions and evaluating its performance across different regression models. This enhances confidence in the reliability and generalizability of the learned latent representations.</p>
<p>The initialization of model parameters can impact the training process and the final performance of the model. Different initializations can lead to the model getting to different local minima, resulting in variable performance. To check the consistency of the performance, we trained both the networks (CAE and vanilla AE) using 4 different initial conditions. By training the model with multiple initial conditions, we can evaluate its robustness and consistency in learning informative latent representations. The <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref> (Days to Pollen) and <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref> (Yield) show a comparison of performance between a vanilla auto-encoder and compositional autoencoder for the traits of &#x2018;Days to Pollen&#x2019; and &#x2018;Yield&#x2019; after performing a 5-fold cross-validation. We clearly see the consistency of prediction accuracy across different model initializations.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Table shows results obtained for Days to Pollen trait using a vanilla autoencoder (AE) and the compositional autoencoder (CAE).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Metric - Model</th>
<th valign="top" align="center">Init. 1</th>
<th valign="top" align="center">Init. 2</th>
<th valign="top" align="center">Init. 3</th>
<th valign="top" align="center">Init. 4</th>
<th valign="top" align="center">ML Model</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> <bold>- CAE</bold>
</td>
<td valign="top" align="center">0.681 (0.04)</td>
<td valign="top" align="center">0.68 (0.035)</td>
<td valign="top" align="center">0.676 (0.033)</td>
<td valign="top" align="center">0.68 (0.034)</td>
<td valign="top" align="center">Xg-Boost Regression</td>
</tr>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> <bold>- AE</bold>
</td>
<td valign="top" align="center">0.08 (0.02)</td>
<td valign="top" align="center">0.127 (0.02)</td>
<td valign="top" align="center">0.108 (0.03)</td>
<td valign="top" align="center">0.110 (0.03)</td>
<td valign="top" align="center">Ridge Regression</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The latent vectors generated using these 2 models performed differently with different ML models and the table below shows the best results among all the models that we tested.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Table shows the results obtained for yield prediction trait using a vanilla autoencoder (AE) and the compositional autoencoder (CAE).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Metric - Model</th>
<th valign="top" align="center">Init. 1</th>
<th valign="top" align="center">Init. 2</th>
<th valign="top" align="center">Init. 3</th>
<th valign="top" align="center">Init. 4</th>
<th valign="top" align="center">ML Model</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> <bold>- CAE</bold>
</td>
<td valign="top" align="center">0.351 (0.058)</td>
<td valign="top" align="center">0.35 (0.054)</td>
<td valign="top" align="center">0.338 (0.058)</td>
<td valign="top" align="center">0.345 (0.06)</td>
<td valign="top" align="center">Xg-Boost Regression</td>
</tr>
<tr>
<td valign="top" align="center">
<italic>R</italic>
<sup>2</sup> <bold>- AE</bold>
</td>
<td valign="top" align="center">0.026 (0.017)</td>
<td valign="top" align="center">0.027 (0.014)</td>
<td valign="top" align="center">0.029 (0.015)</td>
<td valign="top" align="center">0.028 (0.015)</td>
<td valign="top" align="center">Ridge Regression</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The latent vectors generated using these 2 models performed differently with different ML models and the table below shows the best results among all the models that we tested.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>We finally report on varying various hyperparameters of the CAE, and their sensitivity to the downstream performance:</p>
<list list-type="bullet">
<list-item>
<p>Masking: We evaluated the effect of input masking. Input masking improves the robustness and generalization of autoencoders by forcing them to reconstruct missing or corrupted data, which helps the model learn more significant features and patterns. This technique also acts as a regularization method, preventing overfitting and enhancing performance in various downstream tasks. <xref ref-type="table" rid="T9">
<bold>Table&#xa0;9</bold>
</xref> shows the reconstruction accuracy as a function of masking fraction and suggests that 20% masking is a good choice. We also observed that performance on the downstream task also improved by using masking while training. <xref ref-type="table" rid="T10">
<bold>Table&#xa0;10</bold>
</xref> shows <italic>R</italic>
<sup>2</sup> observed for different masking percentages.</p>
</list-item>
<list-item>
<p>Network depth: Network depth is an important hyperparameter to explore because it directly influences the model&#x2019;s capacity to learn complex patterns and hierarchical representations within the data. Deeper networks can capture more intricate features and dependencies, potentially leading to improved performance on complex tasks, but they also require careful tuning to avoid issues such as vanishing gradients and overfitting. We evaluated how performance varied when the CAE network depth was varied. <xref ref-type="table" rid="T11">
<bold>Table&#xa0;11</bold>
</xref> shows the performance observed for different-sized fully connected networks. We can see that the downstream performance is nearly independent of network depth.</p>
</list-item>
<list-item>
<p>Size of the latent representation: We next evaluated how the size/dimension of the latent space affected the downstream trait prediction accuracy. Choosing a higher-dimensional latent space can result in better reconstruction accuracy; however, higher-dimensional latent spaces require larger datasets to avoid overfitting of downstream traits. This suggests a balanced approach in designing the dimensionality of the latent space to balance reconstruction accuracy (which improves with increasing latent space dimensionality) with trait regression accuracy (which improves with decreasing latent space dimensionality).</p>
</list-item>
</list>
<table-wrap id="T9" position="float">
<label>Table&#xa0;9</label>
<caption>
<p>CAE reconstruction accuracy for different masking %.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Percentage Masking</th>
<th valign="top" align="center">Val. Loss</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">0%</td>
<td valign="top" align="center">0.08</td>
</tr>
<tr>
<td valign="top" align="center">20%</td>
<td valign="top" align="center">0.05</td>
</tr>
<tr>
<td valign="top" align="center">50%</td>
<td valign="top" align="center">0.05</td>
</tr>
<tr>
<td valign="top" align="center">70%</td>
<td valign="top" align="center">0.05</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T10" position="float">
<label>Table&#xa0;10</label>
<caption>
<p>Downstream trait prediction accuracy (&#x2018;Days to Pollen&#x2019;) for different masking %.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Percentage Masking</th>
<th valign="top" align="center">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">0%</td>
<td valign="top" align="center">0.749</td>
</tr>
<tr>
<td valign="top" align="center">20%</td>
<td valign="top" align="center">0.757</td>
</tr>
<tr>
<td valign="top" align="center">50%</td>
<td valign="top" align="center">0.756</td>
</tr>
<tr>
<td valign="top" align="center">70%</td>
<td valign="top" align="center">0.763</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T11" position="float">
<label>Table&#xa0;11</label>
<caption>
<p>Table shows the performance observed for &#x2018;Days to Pollen&#x2019; for different sized networks.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">No. Parameters</th>
<th valign="top" align="center">No. Layers</th>
<th valign="top" align="center">CAE - <italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">14.7<italic>M</italic>
</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">0.76</td>
</tr>
<tr>
<td valign="top" align="center">5.5<italic>M</italic>
</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">0.76</td>
</tr>
<tr>
<td valign="top" align="center">2.2<italic>M</italic>
</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">0.76</td>
</tr>
<tr>
<td valign="top" align="center">392<italic>K</italic>
</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">0.76</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We remind the reader that our disentangled latent space is a vector consisting of three sets of components &#x2014; &#x2018;Genotype features.&#x2019; &#x2018;field-level environment features,&#x2019; and &#x2018;plant-level environment features.&#x2019; As the genotype is a common characteristic, we assign more dimensions to capture its effects. Field-level environmental features are allocated fewer dimensions, and plant-level environmental features are given the least. <xref ref-type="table" rid="T12">
<bold>Table&#xa0;12</bold>
</xref> shows how the performance of the downstream regression accuracy varies as the latent dimension is doubled from 10 to 20 to 40 to 80 dimensions. We see an asymptotic behavior after a latent space of 20 dimensions.</p>
<table-wrap id="T12" position="float">
<label>Table&#xa0;12</label>
<caption>
<p>Table shows the performance observed for &#x2018;Days to Pollen&#x2019; for different latent configurations with 2.2 M training parameters.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Latent space dims (Geno-Env-Plant dims)</th>
<th valign="top" align="center">CAE - <italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">10 (6&#x2212;2&#x2212;2)</td>
<td valign="top" align="center">0.69</td>
</tr>
<tr>
<td valign="top" align="center">20 (12&#x2212;4&#x2212;4)</td>
<td valign="top" align="center">0.76</td>
</tr>
<tr>
<td valign="top" align="center">40 (24&#x2212;8&#x2212;8)</td>
<td valign="top" align="center">0.76</td>
</tr>
<tr>
<td valign="top" align="center">80 (48&#x2212;16&#x2212;16)</td>
<td valign="top" align="center">0.77</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusion</title>
<p>This study introduced a novel compositional autoencoder (CAE) framework designed to disentangle genotype-specific and environment-specific features from high-dimensional data, thereby enhancing trait prediction in plant breeding and genetics programs. The CAE effectively separates these intertwined factors by leveraging a hierarchical disentanglement of latent spaces, leading to superior predictive performance for key agricultural traits such as &#x201c;Days to Pollen&#x201d; and &#x201c;Yield.&#x201d; Our results demonstrate that the CAE outperforms traditional methods, including Principal Component Analysis (PCA) and vanilla autoencoders, in capturing relevant information for trait prediction. The evaluation of various network architectures, latent space dimensions, and hyperparameter tuning further validated the robustness and generalizability of the CAE model. Specifically, the CAE showed consistent performance improvements across different initialization conditions and regression models, underscoring its reliability in practical applications.</p>
<p>By effectively disentangling genotype and environment-specific features, the CAE offers a powerful tool for improving the accuracy and reliability of predictive models in agriculture, ultimately contributing to more informed decision-making in breeding programs and agricultural management. Overall, our contributions in this paper are as follows: a) we report a generalized architecture &#x2013; compositional autoencoder (CAE) &#x2013; that can produce a disentangled, low-dimensional, latent representation (that respects hierarchical relationships), given high-dimensional data across a diverse set of plant genotypes. In this case, the effects of genotype and environment on hyperspectral data collected from plants. b) This architecture (CAE) shows an improvement in predicting &#x2018;Days to Pollen&#x2019;, a measure of flowering time which plays a key role in determining crop variety suitability to different environments, when compared to standard vanilla autoencoder or PCA. c) The CAE latent representation produces models with improved accuracy in predicting the trait &#x2018;Yield&#x2019; (i.e. the amount of grain produced by a given crop variety grown on a fixed amount of land), which is both critically important to farmers and considered quite difficult to predict from mid-season sensor measurements when compared to the current state-of-art methods like classical autoencoders.</p>
<p>There are several avenues for future work. First, it will be interesting to explore the viability of compositional autoencoders for making trait predictions using the disentangled GXE features using other sensing modalities <xref ref-type="bibr" rid="B35">Shrestha et&#xa0;al. (2024)</xref> like (a) UAV-based hyperspectral imagery and (b) satellite-based multispectral imagery. Second, applying CAE to time-series high-dimensional data collected on diversity panels can produce disentangled low-dimensional time trajectories that could provide biological insight. Finally, integrating these disentangled latent representations with other data (crop models, physiological measurements) may be a promising approach for creating accurate end-of-season trait prediction models using mid-season data.</p>
<p>We conclude by identifying the following limitations of our work: (a) We evaluated the performance of the CAE on two specific traits that were phenotyped in the field experiments. Our future work will focus on evaluating the CAE on a broader range of traits; (b) Our study is based on hyperspectral reflectance data from a specific maize diversity panel. Our future work is focused on extending this to other datasets and environments; (c) While we demonstrate the technical advantages of disentanglement, it is not immediately clear how to connect these disentangled features to biological insights.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <uri xlink:href="https://figshare.com/articles/dataset/Hyperspectral_reflectance_data_molecular_and_weights_for_trained_model/24808491/4">https://figshare.com/articles/dataset/Hyperspectral_reflectance_data_molecular_and_weights_for_trained_model/24808491/4</uri>; <uri xlink:href="https://bitbucket.org/baskargroup/cae_hyperspectral/src/main/">https://bitbucket.org/baskargroup/cae_hyperspectral/src/main/</uri>.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>AP: Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. TJ: Conceptualization, Investigation, Software, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. MT: Data curation, Formal analysis, Writing &#x2013; review &amp; editing. JS: Data curation, Project administration, Writing &#x2013; review &amp; editing. BG: Conceptualization, Project administration, Resources, Supervision, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the AI Institute for Resilient Agriculture (USDA-NIFA 2021-67021-35329) and Iowa State University Plant Science Institute.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2024.1476070/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2024.1476070/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adak</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Murray</surname> <given-names>S. C.</given-names>
</name>
<name>
<surname>Anderson</surname> <given-names>S. L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Temporal phenomic predictions from unoccupied aerial systems can outperform genomic predictions</article-title>. <source>G3</source> <volume>13</volume>, <fpage>jkac294</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/g3journal/jkac294</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aguate</surname> <given-names>F. M.</given-names>
</name>
<name>
<surname>Trachsel</surname> <given-names>S.</given-names>
</name>
<name>
<surname>P&#xe9;rez</surname> <given-names>L. G.</given-names>
</name>
<name>
<surname>Burgue&#xf1;o</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Crossa</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Balzarini</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Use of hyperspectral image data outperforms vegetation indices in prediction of maize yield</article-title>. <source>Crop Sci.</source> <volume>57</volume>, <fpage>2517</fpage>&#x2013;<lpage>2524</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2135/cropsci2017.01.0007</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alexander</surname> <given-names>T. A.</given-names>
</name>
<name>
<surname>Irizarry</surname> <given-names>R. A.</given-names>
</name>
<name>
<surname>Bravo</surname> <given-names>H. C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Capturing discrete latent structures: choose LDs over PCs</article-title>. <source>Biostatistics</source> <volume>24</volume>, <fpage>1</fpage>&#x2013;<lpage>16</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/biostatistics/kxab030</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bachmann</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Mizrahi</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Atanov</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zamir</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multimae: Multi-modal multi-task masked autoencoders</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:2204.01678</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bai</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). &#x201c;<article-title>Hierarchical graph convolutional skeleton transformer for action recognition</article-title>,&#x201d; in <conf-name>2022 IEEE International Conference on Multimedia and Expo (ICME)</conf-name>. <fpage>01</fpage>&#x2013;<lpage>06</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICME52920.2022.9859781</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baio</surname> <given-names>F. H. R.</given-names>
</name>
<name>
<surname>Santana</surname> <given-names>D. C.</given-names>
</name>
<name>
<surname>Teodoro</surname> <given-names>L. P. R.</given-names>
</name>
<name>
<surname>Oliveira</surname> <given-names>I. C.</given-names>
</name>
<name>
<surname>Gava</surname> <given-names>R.</given-names>
</name>
<name>
<surname>de Oliveira</surname> <given-names>J. L. G.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Maize yield prediction with machine learning, spectral variables and irrigation management</article-title>. <source>Remote Sens.</source> <volume>15</volume>, <fpage>79</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs15010079</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Banar</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Bryan-Kinns</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Colton</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A tool for generating controllable variations of musical themes using variational autoencoders with latent space regularisation</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>37</volume>, <fpage>16401</fpage>&#x2013;<lpage>16403</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1609/aaai.v37i13.27059</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Burgess</surname> <given-names>C. P.</given-names>
</name>
<name>
<surname>Higgins</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Pal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Matthey</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Watters</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Desjardins</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Understanding disentangling in <italic>&#x3b2;</italic>-vae</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:1804.03599</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cha</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Thiyagalingam</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Orthogonality-enforced latent space in autoencoders: An approach to learning disentangled representations</article-title>,&#x201d; in <conf-name>Proceedings of the 40th International Conference on Machine Learning</conf-name>, Vol. <volume>202</volume>, <fpage>3913</fpage>&#x2013;<lpage>3948</lpage> (<publisher-name>Proceedings of Machine Learning Research</publisher-name>).</citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>On hierarchical disentanglement of interactive behaviors for multimodal spatiotemporal data with incompleteness</article-title>,&#x201d; in <conf-name>Proceedings of the 29th ACM SIGKDD Conference on Knowledge Discovery and Data Mining</conf-name>. <fpage>213</fpage>&#x2013;<lpage>225</lpage> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3580305.3599448</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Comas</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ghimire</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Sznaier</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Camps</surname> <given-names>O.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Self-supervised decomposition, disentanglement and prediction of video sequences while interpreting dynamics: A koopman perspective</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:2110.00547</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cui</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Fukumoto</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Suzuki</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Tomuro</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Enhanced coherence-aware network with hierarchical disentanglement for aspect-category sentiment analysis</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:2403.10214</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>de Leon</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Kaeppler</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Lima</surname> <given-names>D. C.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Estimation of maize yield and flowering time using multi-temporal uav-based hyperspectral data</article-title>. <source>Remote Sens.</source> <volume>14 (13)</volume>, 3052. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs14133052</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feldmann</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Gage</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Turner-Hissong</surname> <given-names>S. D.</given-names>
</name>
<name>
<surname>Ubbens</surname> <given-names>J. R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Images carried before the fire: The power, promise, and responsibility of latent phenotyping in plants</article-title>. <source>Plant Phenome J.</source> <volume>4</volume>, <elocation-id>e20023</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/ppj2.20023</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gage</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Richards</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Lepak</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Kaczmar</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Soman</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chowdhary</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>In-field whole-plant maize architecture characterized by subcanopy rovers and latent space phenotyping</article-title>. <source>Plant Phenome J.</source> <volume>2</volume>, <fpage>190011</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2135/tppj2019.07.0011</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gomari</surname> <given-names>D. P.</given-names>
</name>
<name>
<surname>Schweickart</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Cerchietti</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Paietta</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Fernandez</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Al-Amin</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Variational autoencoders learn transferrable representations of metabolomics data</article-title>. <source>Commun. Biol.</source> <volume>5</volume>, <fpage>645</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42003-022-03579-3</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guil</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Hortal</surname> <given-names>J.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Moreno</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Machordom</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Effects of macro and micro-environmental factors on the species richness of terrestrial tardigrade assemblages in an iberian mountain environment</article-title>. <source>Landscape Ecol.</source> <volume>24</volume>, <fpage>375</fpage>&#x2013;<lpage>390</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10980-008-9312-x</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Higgins</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Matthey</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Pal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Burgess</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Glorot</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Botvinick</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). &#x201c;<article-title>beta-VAE: Learning basic visual concepts with a constrained variational framework</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations</conf-name>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iwasaki</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Cooray</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Takeuchi</surname> <given-names>T. T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Extracting an informative latent representation of high-dimensional galaxy spectra</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:2311.17414</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jia</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Salzmann</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Darrell</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2010</year>). <source>Factorized latent spaces with structured sparsity</source>. (New York, USA: Curran Associates, Inc.), <fpage>982</fpage>&#x2013;<lpage>990</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Tross</surname> <given-names>M. C.</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Newton</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Mural</surname> <given-names>R. V.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Imitating the &#x201c;breeder&#x2019;s eye&#x201d;: Predicting grain yield from measurements of non-yield traits</article-title>. <source>Plant Phenome J.</source> <volume>7</volume>, <elocation-id>e20102</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/ppj2.20102</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaleita</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Steward</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ewing</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Westgate</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hatfield</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ashlock</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Novel analysis of hyperspectral reflectance data for detecting onset of pollen shed in maize</article-title>. <source>Trans. ASABE</source> <volume>49</volume>, 1947&#x2013;1954. doi:&#xa0;<pub-id pub-id-type="doi">10.13031/2013.22274</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karras</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Laine</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Aila</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A style-based generator architecture for generative adversarial networks</article-title>. <source>arXiv</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR41558.2019</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Mnih</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Disentangling by factorising</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:1802.05983</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname> <given-names>D. P.</given-names>
</name>
<name>
<surname>Rezende</surname> <given-names>D. J.</given-names>
</name>
<name>
<surname>Mohamed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Welling</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Semi-supervised learning with deep generative models</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:1406.5298</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname> <given-names>D. P.</given-names>
</name>
<name>
<surname>Welling</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An introduction to variational autoencoders</article-title>. <source>Foundations Trends&#xae; Mach. Learn.</source> <volume>12</volume>, <fpage>307</fpage>&#x2013;<lpage>392</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1561/2200000056</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname> <given-names>D. P.</given-names>
</name>
<name>
<surname>Welling</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Auto-encoding variational bayes</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:1312.6114</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kopf</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Claassen</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Latent representation learning in biology and translational medicine</article-title>. <source>Patterns (N. Y.)</source> <volume>2</volume>, <fpage>100198</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.patter.2021.100198</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kulkarni</surname> <given-names>T. D.</given-names>
</name>
<name>
<surname>Whitney</surname> <given-names>W. F.</given-names>
</name>
<name>
<surname>Kohli</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Tenenbaum</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Deep convolutional inverse graphics network</article-title>,&#x201d; in <source>Advances in neural information processing systems</source>, vol. <volume>28</volume> . Eds. <person-group person-group-type="editor">
<name>
<surname>Cortes</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Lawrence</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Sugiyama</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Garnett</surname> <given-names>R.</given-names>
</name>
</person-group> (<publisher-name>Curran Associates, Inc</publisher-name>).</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>J.-C.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Ling</surname> <given-names>Z.-H.</given-names>
</name>
<name>
<surname>Su</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Dialbert: A hierarchical pre-trained model for conversation disentanglement</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:2004.03760</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Towards disentangling latent space for unsupervised semantic face editing</article-title>. <source>IEEE Trans. Image Process.</source> <volume>31</volume>, <fpage>1475</fpage>&#x2013;<lpage>1489</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2022.3142527</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mazaheri</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Heckwolf</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vaillancourt</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Gage</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Burdo</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Heckwolf</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Genome-wide association analysis of stalk biomass and anatomical traits in maize</article-title>. <source>BMC Plant Biol.</source> <volume>19</volume>, <fpage>1</fpage>&#x2013;<lpage>17</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12870-019-1653-x</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mural</surname> <given-names>R. V.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Grzybowski</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Tross</surname> <given-names>M. C.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Association mapping across a multitude of traits collected in diverse environments in maize</article-title>. <source>GigaScience</source> <volume>11</volume>, <fpage>giac080</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/gigascience/giac080</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Disentangling the latent space of GANs for semantic face editing</article-title>. <source>PloS One</source> <volume>18</volume>, <elocation-id>e0293496</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0293496</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shrestha</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Powadi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Davis</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ayanlade</surname> <given-names>T. T.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>H.-y.</given-names>
</name>
<name>
<surname>Tross</surname> <given-names>M. C.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Plot-level satellite imagery can substitute for uavs in assessing maize phenotypes across multistate field trials</article-title>. <source>agriRxiv</source>, <fpage>20240201322</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.31220/agriRxiv.2024.00251</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>M. K.</given-names>
</name>
<name>
<surname>Niaz</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>K. N.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Image generation model applying pca on latent space</article-title>,&#x201d; in <conf-name>Proceedings of the 2023 2nd Asia Conference on Algorithms, Computing and Machine Learning</conf-name>. <fpage>419</fpage>&#x2013;<lpage>423</lpage> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3590003.3590080</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Unsupervised orthogonal facial representation extraction via image reconstruction with correlation minimization</article-title>. <source>Neurocomputing</source> <volume>337</volume>, <fpage>203</fpage>&#x2013;<lpage>217</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neucom.2019.01.068</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Weiss</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Fully-hierarchical fine-grained prosody modeling for interpretable speech synthesis</article-title>,&#x201d; in <conf-name>ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)</conf-name>. <fpage>6264</fpage>&#x2013;<lpage>6268</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICASSP40776.2020.9053520</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tatro</surname> <given-names>N. J.</given-names>
</name>
<name>
<surname>Das</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>P.-Y.</given-names>
</name>
<name>
<surname>Chenthamarakshan</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Pro{gae}: A geometric autoencoder-based generative model for disentangling protein conformational space</article-title>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tross</surname> <given-names>M. C.</given-names>
</name>
<name>
<surname>Grzybowski</surname> <given-names>M. W.</given-names>
</name>
<name>
<surname>Jubery</surname> <given-names>T. Z.</given-names>
</name>
<name>
<surname>Grove</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Nishimwe</surname> <given-names>A. V.</given-names>
</name>
<name>
<surname>Torres-Rodriguez</surname> <given-names>J. V.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Data driven discovery and quantification of hyperspectral leaf reflectance phenotypes across a maize diversity panel</article-title>. <source>Plant Phenome J.</source> <volume>7 (1)</volume>, <fpage>e20106</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2023.12.15.571950</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ubbens</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cieslak</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Prusinkiewicz</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Parkin</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Ebersbach</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Stavness</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Latent space phenotyping: Automatic image-based phenotyping for treatment studies</article-title>. <source>Plant Phenomics</source> <volume>2020</volume>, 5801869. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/2020/5801869</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Advanced conditional variational autoencoders (a-cvae): Towards interpreting open-domain conversation generation via disentangling latent feature representation</article-title>. <source>arXiv</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.21203/rs.3.rs-1845437/v1</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Watters</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Matthey</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Burgess</surname> <given-names>C. P.</given-names>
</name>
<name>
<surname>Lerchner</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Spatial broadcast decoder: A simple architecture for learning disentangled representations in vaes</article-title>. <source>arXiv</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv:1901.07017</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weber</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Araus</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cairns</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Sanchez</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Melchinger</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Orsini</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Prediction of grain yield using reflectance spectra of canopy and leaves in maize plants grown under different water regimes</article-title>. <source>Field Crops Res.</source> <volume>128</volume>, <fpage>82</fpage>&#x2013;<lpage>90</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.fcr.2011.12.016</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wei</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Controlling facial attribute synthesis by disentangling attribute feature axes in latent space</article-title>,&#x201d; in <conf-name>2023 IEEE International Conference on Image Processing (ICIP)</conf-name>. <fpage>346</fpage>&#x2013;<lpage>350</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICIP49359.2023.10223056</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Arildsen</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>Z.-H.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Improved disentangled speech representations using contrastive learning in factorized hierarchical variational autoencoder</article-title>,&#x201d; in <conf-name>2023 31st European Signal Processing Conference (EUSIPCO)</conf-name>. <fpage>1330</fpage>&#x2013;<lpage>1334</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.23919/EUSIPCO58844.2023.10289926</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>T.</given-names>
</name>
<name>
<surname>McGuinness</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Gurrin</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Learning multiple views with orthogonal denoising autoencoders</article-title>. <source>Lect. Notes Comput. Sci.</source> <volume>9516</volume>, <page-range>313&#x2013;324</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-27671-726</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yendrek</surname> <given-names>C. R.</given-names>
</name>
<name>
<surname>Tomaz</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Montes</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Morse</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>P. J.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>High-throughput phenotyping of maize leaf physiological and biochemical traits using hyperspectral reflectance</article-title>. <source>Plant Physiol.</source> <volume>173</volume>, <fpage>614</fpage>&#x2013;<lpage>626</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1104/pp.16.01447</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Lv</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Hyperspectral imaging combined with cnn for maize variety identification</article-title>. <source>Front. Plant Sci.</source> <volume>14</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2023.1254548</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Disentangling latent space for vae by label relevant/irrelevant dimensions</article-title>,&#x201d; in <conf-name>2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>. <fpage>12184</fpage>&#x2013;<lpage>12193</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2019.01247</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.-N.</given-names>
</name>
<name>
<surname>Ling</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>An overview on data representation learning: From traditional feature learning to recent deep learning</article-title>. <source>J. Finance Data Sci.</source> <volume>2</volume>, <fpage>265</fpage>&#x2013;<lpage>278</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jfds.2017.05.001</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>