<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2022.845524</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Automated Machine Learning: A Case Study of Genomic &#x201C;Image-Based&#x201D; Prediction in Maize Hybrids</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Galli</surname>
<given-names>Giovanni</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sabadin</surname>
<given-names>Felipe</given-names>
</name>
<xref rid="aff2" ref-type="aff"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1346964/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yassue</surname>
<given-names>Rafael Massahiro</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Galves</surname>
<given-names>Cassia</given-names>
</name>
<xref rid="aff3" ref-type="aff"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Carvalho</surname>
<given-names>Humberto Fanelli</given-names>
</name>
<xref rid="aff4" ref-type="aff"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Crossa</surname>
<given-names>Jose</given-names>
</name>
<xref rid="aff5" ref-type="aff"><sup>5</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/50360/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Montesinos-L&#x00F3;pez</surname>
<given-names>Osval Antonio</given-names>
</name>
<xref rid="aff6" ref-type="aff"><sup>6</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/988922/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Fritsche-Neto</surname>
<given-names>Roberto</given-names>
</name>
<xref rid="aff1" ref-type="aff"><sup>1</sup></xref>
<xref rid="aff7" ref-type="aff"><sup>7</sup></xref>
<xref rid="c001" ref-type="corresp"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/453020/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Genetics, Luiz de Queiroz College of Agriculture, University of S&#x00E3;o Paulo</institution>, <addr-line>Piracicaba</addr-line>, <country>Brazil</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Plant and Environmental Sciences, Virginia Tech</institution>, <addr-line>Blacksburg, VA</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Food Engineering, University of Saskatchewan</institution>, <addr-line>Saskatoon, SK</addr-line>, <country>Canada</country></aff>
<aff id="aff4"><sup>4</sup><institution>Centro de Biotecnolog&#x00ED;a y Gen&#x00F3;mica de Plantas, Universidad Polit&#x00E9;cnica de Madrid</institution>, <addr-line>Madrid</addr-line>, <country>Spain</country></aff>
<aff id="aff5"><sup>5</sup><institution>International Maize and Wheat Improvement Center (CIMMYT)</institution>, <addr-line>Texcoco</addr-line>, <country>Mexico</country></aff>
<aff id="aff6"><sup>6</sup><institution>Facultad de Telem&#x00E1;tica, Universidad de Colima</institution>, <addr-line>Colima</addr-line>, <country>Mexico</country></aff>
<aff id="aff7"><sup>7</sup><institution>International Rice Research Institute (IRRI)</institution>, <addr-line>Los Ba&#x00F1;os</addr-line>, <country>Philippines</country></aff>
<author-notes>
<fn id="fn0001" fn-type="edited-by"><p>Edited by: Aalt-Jan Van Dijk, Wageningen University and Research, Netherlands</p></fn>
<fn id="fn0002" fn-type="edited-by"><p>Reviewed by: Dong Xu, University of Missouri, United States; Guillaume Ramstein, Cornell University, United States</p></fn>
<corresp id="c001">&#x002A;Correspondence: Roberto Fritsche-Neto, <email>r.fritscheneto@irri.org</email></corresp>
<fn id="fn0003" fn-type="other"><p>This article was submitted to Technical Advances in Plant Science, a section of the journal Frontiers in Plant Science</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>07</day>
<month>03</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>13</volume>
<elocation-id>845524</elocation-id>
<history>
<date date-type="received">
<day>29</day>
<month>12</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>02</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2022 Galli, Sabadin, Yassue, Galves, Carvalho, Crossa, Montesinos-L&#x00F3;pez and Fritsche-Neto.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Galli, Sabadin, Yassue, Galves, Carvalho, Crossa, Montesinos-L&#x00F3;pez and Fritsche-Neto</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Machine learning methods such as multilayer perceptrons (MLP) and Convolutional Neural Networks (CNN) have emerged as promising methods for genomic prediction (GP). In this context, we assess the performance of MLP and CNN on regression and classification tasks in a case study with maize hybrids. The genomic information was provided to the MLP as a relationship matrix and to the CNN as &#x201C;genomic images.&#x201D; In the regression task, the machine learning models were compared along with GBLUP. Under the classification task, MLP and CNN were compared. In this case, the traits (plant height and grain yield) were discretized in such a way to create balanced (moderate selection intensity) and unbalanced (extreme selection intensity) datasets for further evaluations. An automatic hyperparameter search for MLP and CNN was performed, and the best models were reported. For both task types, several metrics were calculated under a validation scheme to assess the effect of the prediction method and other variables. Overall, MLP and CNN presented competitive results to GBLUP. Also, we bring new insights on automated machine learning for genomic prediction and its implications to plant breeding.</p>
</abstract>
<kwd-group>
<kwd>non-image to image</kwd>
<kwd>multilayer perceptrons</kwd>
<kwd>convolutional neural networks</kwd>
<kwd>AutoML</kwd>
<kwd>accuracy</kwd>
</kwd-group>
<contract-num rid="cn1">INV-003439 BMGF/FCDO</contract-num>
<contract-sponsor id="cn1">The Bill and Melinga Gates Foundation (BMGF)</contract-sponsor>
<counts>
<fig-count count="1"/>
<table-count count="2"/>
<equation-count count="2"/>
<ref-count count="48"/>
<page-count count="13"/>
<word-count count="10524"/>
</counts>
</article-meta>
</front>
<body>
<sec id="sec1" sec-type="intro">
<title>Introduction</title>
<p>Genomic prediction (GP) arose as a breeding tool capable of enabling a considerable increase in the rates of genetic gain. In this context, three decades of scientific research have shown that the accuracy of this statistical approach might be conditioned to a series of factors, including the quality and pre-processing of the phenotypic data (<xref ref-type="bibr" rid="ref18">Galli et al., 2018</xref>), the platform used to obtain genomic information and how it is processed (<xref ref-type="bibr" rid="ref21">Granato et al., 2018</xref>; <xref ref-type="bibr" rid="ref40">Sousa et al., 2019</xref>), the population mating design (<xref ref-type="bibr" rid="ref14">Fritsche-Neto et al., 2018</xref>), the intrinsic genetic architecture of the trait (<xref ref-type="bibr" rid="ref3">Alves et al., 2019</xref>), the genetic structure of the population (<xref ref-type="bibr" rid="ref26">Lyra et al., 2018</xref>), how the genotype-by-environment interaction is dealt with (<xref ref-type="bibr" rid="ref2">Alves et al., 2021</xref>; <xref ref-type="bibr" rid="ref10">Costa-Neto et al., 2021</xref>), and which prediction methods are used [e.g., BayesA, BayesB (<xref ref-type="bibr" rid="ref29">Meuwissen et al., 2001</xref>); GBLUP (<xref ref-type="bibr" rid="ref7">Bernardo, 1994</xref>; <xref ref-type="bibr" rid="ref46">VanRaden, 2008</xref>); Reproducing Kernel Hilbert Spaces (<xref ref-type="bibr" rid="ref11">de Los Campos et al., 2009</xref>)].</p>
<p>Several statistical machine learning methods have been adopted for GP because they can help improve genome-enabled prediction accuracy since they are able to make computers learn models or patterns that could be used for analysis, interpretation, prediction, and decision-making. For example, Random Forest (<xref ref-type="bibr" rid="ref33">Montesinos-L&#x00F3;pez et al., 2021a</xref>), Support Vector Machine (<xref ref-type="bibr" rid="ref30">Montesinos-L&#x00F3;pez et al., 2019</xref>), and Gradient Boosting Machine (<xref ref-type="bibr" rid="ref34">Montesinos-L&#x00F3;pez et al., 2021b</xref>). Recently neural networks have been intensively studied and applied in genome-based breeding (<xref ref-type="bibr" rid="ref30">Montesinos-L&#x00F3;pez et al., 2019</xref>, <xref ref-type="bibr" rid="ref34">2021b</xref>). However, one reason why so many types of statistical machine learning methods have been implemented in GP is that no universal best prediction model can be used under all circumstances.</p>
<p>Multilayer perceptrons (MLPs; fully connected layers) and Convolutional Neural Networks (CNNs; fully connected layers and convolutional/pooling filters) are two common types of neural networks (NN). These methods are characterized by the sequentially stacking (several) layers, which automatically identifies latent patterns or features from data (<xref ref-type="bibr" rid="ref42">Trevisan et al., 2020</xref>). For a technically accurate and contextualized explanation of such models, refer to <xref ref-type="bibr" rid="ref37">P&#x00E9;rez-Enciso and Zingaretti (2019)</xref>. This rising interest is fundamentally associated with the increasing availability of computational power (e.g., graphical processing unit computing, cloud computation, web servers); its success in diverse tasks (such as self-driving vehicles, object detection, and context recognition); ability to work on both regression and classification problems; and especially due to the lower-level restrictions compared to standard models. Also, because neural networks do not need highly pre-processed inputs since these methods are powerful for working directly with raw data (e.g., images, text), and for this reason, they require less human intervention to process data, allowing us to scale machine learning in more interesting ways. For instance, neural networks can perform predictions without restrictive model assumptions; in the context of genetic studies, it does not require specifying the distribution of variables, priors, and the nature of genetic effects (additive, dominance, and epistasis), being theoretically capable of self-adjusting to the underlying genetic architecture (<xref ref-type="bibr" rid="ref37">P&#x00E9;rez-Enciso and Zingaretti, 2019</xref>).</p>
<p>Initial reports suggest that neural networks can compete with the standard GP methods (e.g., GBLUP) in prediction accuracy. Nevertheless, results are highly inconsistent on this matter (<xref ref-type="bibr" rid="ref6">Bellot et al., 2018</xref>; <xref ref-type="bibr" rid="ref27">Ma et al., 2018</xref>; <xref ref-type="bibr" rid="ref31">Montesinos-L&#x00F3;pez et al., 2018a</xref>,<xref ref-type="bibr" rid="ref32">b</xref>; <xref ref-type="bibr" rid="ref4">Azodi et al., 2019</xref>; <xref ref-type="bibr" rid="ref1">Abdollahi-Arpanahi et al., 2020</xref>), and its best use and performance is still to be determined on a broader and most representative spectrum of prediction scenarios. In this context, one of the major challenges for applying this methodology is identifying adequate model structures and hyperparameters (<xref ref-type="bibr" rid="ref6">Bellot et al., 2018</xref>; <xref ref-type="bibr" rid="ref37">P&#x00E9;rez-Enciso and Zingaretti, 2019</xref>; <xref ref-type="bibr" rid="ref45">van Dijk et al., 2021</xref>). Hereon, we refer to hyperparameter as those not learned with the machine learning algorithm but provided by the user before the learning process of the learnable parameters start, e.g., number of hidden layers, number of neurons per layer, learning rate, filter type, and number, activation function, optimization algorithm, regularization type, etc. Since it is an exceptionally flexible algorithm, there is an infinite number of possible configurations. Therefore, automated procedures are required to explore the possibilities and increase the chance of finding near-to-optimal hyperparameters.</p>
<p>NN&#x2019;s calibration and training process are very challenging because many hyperparameters need to be selected, and the adequate selection is time-consuming, cumbersome, and complicated. Automated Machine Learning (AutoML) has great potential for identifying adequate network structures and hyperparameters for a given task (<xref ref-type="bibr" rid="ref23">Jin et al., 2019</xref>). These procedures circumvent hand-designing and testing hyperparameters to save time and effort. Numerous platforms have been developed, such as Auto-sklearn (<xref ref-type="bibr" rid="ref13">Feurer et al., 2015</xref>), Auto-Weka (<xref ref-type="bibr" rid="ref24">Kotthoff et al., 2017</xref>), and AutoKeras (<xref ref-type="bibr" rid="ref23">Jin et al., 2019</xref>); each one with its search algorithm. A comprehensive guide and benchmarking study on the most common search platforms is presented by <xref ref-type="bibr" rid="ref43">Truong et al. (2019)</xref> for further reference. Despite the importance of hyperparameter tuning and the availability of easy-to-use AutoML tools, the number of reports on its use for identifying artificial neural networks for GP is still very limited (<xref ref-type="bibr" rid="ref48">Zingaretti et al., 2020</xref>).</p>
<p>Besides adequate hyperparameter tuning, the performance of a neural network is also determined by the quality and preparation of the data fed for training. For example, in neural network-based GP models, the genomic information has been provided as a genomic relationship/distance matrix (<xref ref-type="bibr" rid="ref31">Montesinos-L&#x00F3;pez et al., 2018a</xref>,<xref ref-type="bibr" rid="ref32">b</xref>), or as the genomic matrix (<xref ref-type="bibr" rid="ref4">Azodi et al., 2019</xref>; <xref ref-type="bibr" rid="ref1">Abdollahi-Arpanahi et al., 2020</xref>). In the case of CNN, the organization of the matrix is meaningful and might contain valuable information (<xref ref-type="bibr" rid="ref37">P&#x00E9;rez-Enciso and Zingaretti, 2019</xref>). For example, <xref ref-type="bibr" rid="ref1">Abdollahi-Arpanahi et al. (2020)</xref> applied CNN with genomic matrices to exploit linkage disequilibrium (LD) patterns between genetic markers. In this case, meaningful filter movements were restricted to a single direction (chromosome-wise), seizing physical linkage disequilibrium (e.g., neighboring markers). Nevertheless, LD is known to vary across the genome (<xref ref-type="bibr" rid="ref6">Bellot et al., 2018</xref>); hence, further advancements to this methodology have been proposed, such as using local convolutional layers applying region-specific filters (<xref ref-type="bibr" rid="ref38">Pook et al., 2020</xref>).</p>
<p>Recently, a work by <xref ref-type="bibr" rid="ref39">Sharma et al. (2019)</xref> has shown the possibility of transforming non-image data (e.g., a genomic matrix) into &#x201C;images&#x201D; (2 or 3-dimensional visual matrices) leveraging dimensionality reduction techniques. In images, data is coherently distributed along with a space pattern, meaning that neighboring pixels share information in all directions, are correlated (<xref ref-type="bibr" rid="ref39">Sharma et al., 2019</xref>). The authors reported the superiority of image-based CNN over original data in machine learning tasks and named the pipeline <italic>DeepInsight</italic>. In context, using genomic images would presumably unlock the potential of CNNs for GP, capturing the relationships between markers over new dimensions.</p>
<p>To test new methodologies in a GP context, a key component of comprehensive and meaningful benchmarking starts with an adequate choice of comparison metrics. In this context, regression tasks have mainly relied on metrics such as Pearson&#x2019;s product&#x2013;moment correlation and its variations (e.g., divided by the trait&#x2019;s heritability), Spearman&#x2019;s correlation, repeatability/heritability, reliability, etc. However, some of these metrics cannot represent the core practice of plant breeding, ranking and selection (<xref ref-type="bibr" rid="ref36">Ornella et al., 2014</xref>; <xref ref-type="bibr" rid="ref8">Blondel et al., 2015</xref>). This problem has been tackled using selection-centered metrics, such as selection coincidence (<xref ref-type="bibr" rid="ref28">Matias et al., 2017</xref>; <xref ref-type="bibr" rid="ref18">Galli et al., 2018</xref>; <xref ref-type="bibr" rid="ref3">Alves et al., 2019</xref>). We add to this matter by unifying ranking and selection by discretizing continuous data and conducting prediction based on classification methods suggested by <xref ref-type="bibr" rid="ref36">Ornella et al. (2014)</xref>. This opens the possibility of comparing methods with a new realm of metrics that better align with the context of plant breeding.</p>
<p>AutoML has a full, yet to be determined, potential application for breeding targeted GP. In this context, we present a comprehensive study on using these technologies for predicting plant height (PH) and grain yield (GY) in maize. The objectives of this research were to: (i) assess the comparative performance of MLP and CNN with the standard model GBLUP at predicting PH and GY in maize; (ii) evaluate the performance of neural networks for the GP of PH and GY in maize in regression and classification contexts using MLP and CNN; (iii) elaborate on the use of AutoML to identify the best hyperparameters to perform GP; (iv) and verify the usefulness transforming genomic information into images for CNN-based GP.</p>
</sec>
<sec id="sec2" sec-type="materials|methods">
<title>Materials and Methods</title>
<sec id="sec3">
<title>Dependent Variables</title>
<sec id="sec4">
<title>Field Trials</title>
<p>The genetic material was composed of 904 maize single-cross hybrids obtained from a partial diallel of 49 tropical inbred lines (<xref ref-type="bibr" rid="ref15">Fritsche-Neto et al., 2019</xref>). Thorough populational description and statistics have been reported on the inbred lines and hybrids (<xref ref-type="bibr" rid="ref14">Fritsche-Neto et al., 2018</xref>; <xref ref-type="bibr" rid="ref3">Alves et al., 2019</xref>; <xref ref-type="bibr" rid="ref16">Galli et al., 2020a</xref>).</p>
<p>The genotypes were arranged in unreplicated trials with the augmented block scheme. Each incomplete block was composed of 18 treatments, 16 regular and two checks (common genotypes). The trials were carried out at Piracicaba-S&#x00E3;o Paulo (22&#x00B0;42&#x2032;23&#x2033; S, 47&#x00B0;38&#x2032;14&#x2033; W, 535&#x2009;m) and Anhembi-S&#x00E3;o Paulo (22&#x00B0;50&#x2032;51&#x2033; S, 48&#x00B0;01&#x2032;06&#x2033; W, 466&#x2009;m), during the second growing seasons of 2016 (738 hybrids) and 2017 (789 hybrids), under two nitrogen application regimes (ideal: 0.1&#x2009;Mg&#x2009;ha<sup>&#x2212;1</sup> and low: 0.03&#x2009;Mg&#x2009;ha<sup>&#x2212;1</sup>). Each experimental unit was composed of a 7&#x2009;m row. The single-crosses were phenotyped for GY (Mg&#x2009;ha<sup>&#x2212;1</sup>) and PH (cm). GY was estimated as the production of a plot corrected for 13% moisture. PH was obtained as the mean height, measured from soil to flag leaf, of five plants in the plot.</p>
</sec>
<sec id="sec5">
<title>Phenotypic Analysis</title>
<p>The genotypic values of hybrids were obtained with a joint linear mixed model using in ASReml-R (<xref ref-type="bibr" rid="ref19">Gilmour et al., 2009</xref>) following:</p>
<disp-formula id="E1">
<mml:math id="M1">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>&#x03B2;</mml:mi>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mi>b</mml:mi>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mstyle mathvariant="bold">
<mml:mn>2</mml:mn>
</mml:mstyle>
</mml:msub>
<mml:mi>g</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>&#x03B5;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M2">
<mml:mi>y</mml:mi>
</mml:math>
</inline-formula> is the phenotype (PH or GY); <inline-formula>
<mml:math id="M3">
<mml:mi>&#x03B2;</mml:mi>
</mml:math>
</inline-formula> is the vector of fixed effects of check, environment (combinations of site, year, and nitrogen regime) and check <inline-formula>
<mml:math id="M4">
<mml:mo>&#x00D7;</mml:mo>
</mml:math>
</inline-formula> environment; <inline-formula>
<mml:math id="M5">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>~</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x0399;</mml:mi>
<mml:msubsup>
<mml:mi>&#x03C3;</mml:mi>
<mml:mi>b</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the random effect of block-within-environment; <inline-formula>
<mml:math id="M6">
<mml:mi>g</mml:mi>
</mml:math>
</inline-formula>
<inline-formula>
<mml:math id="M7">
<mml:mrow>
<mml:mo>~</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x0399;</mml:mi>
<mml:msubsup>
<mml:mi>&#x03C3;</mml:mi>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the random effect of regular genotypes (genotypic values); and <inline-formula>
<mml:math id="M8">
<mml:mrow>
<mml:mi>&#x03B5;</mml:mi>
<mml:mo>~</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mstyle mathvariant="bold">
<mml:mn>0</mml:mn>
</mml:mstyle>
<mml:mi mathvariant="normal">, </mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msubsup>
<mml:mi>&#x03C3;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msubsup>
<mml:mi>&#x03C3;</mml:mi>
<mml:mn>2</mml:mn>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mo>&#x2026;</mml:mo>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msubsup>
<mml:mi>&#x03C3;</mml:mi>
<mml:mn>8</mml:mn>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is a vector of residuals structured by environment estimated from the common treatments (checks). <inline-formula>
<mml:math id="M9">
<mml:mi mathvariant="normal">X</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M10">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">Z</mml:mi>
<mml:mstyle mathvariant="bold">
<mml:mn>1</mml:mn>
</mml:mstyle>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math id="M11">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">Z</mml:mi>
<mml:mstyle mathvariant="bold">
<mml:mn>2</mml:mn>
</mml:mstyle>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the incidence matrices of the mentioned factors. Likelihood Ratio Test (LRT) was used to determine the significance of random effects.</p>
<p>Additionally, a similar model was fit, having check as fixed and regular genotypes, environment, genotype (checks) <inline-formula>
<mml:math id="M12">
<mml:mo>&#x00D7;</mml:mo>
</mml:math>
</inline-formula> environment, and block-within-environment as random for the estimation of variance components. Repeatability at plot level <inline-formula>
<mml:math id="M13">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>/</mml:mo>
<mml:mfenced>
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mi>&#x03B5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> was estimated having <inline-formula>
<mml:math id="M14">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M15">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math id="M16">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mi>&#x03B5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> as the genotypic (hybrids), genotypic (checks) <inline-formula>
<mml:math id="M17">
<mml:mo>&#x00D7;</mml:mo>
</mml:math>
</inline-formula> environment, and residual variances, respectively. The residual variance (<inline-formula>
<mml:math id="M18">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mi>&#x03B5;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) was regarded as the mean residual across environments.</p>
</sec>
<sec id="sec6">
<title>Genotypic Values Pre-processing</title>
<p>The ultimate goal of plant breeding is ranking and selecting the best genotypes. A common practice is categorizing genotypes in selected and non-selected based on their genetic merit. In this context, the subsequent analysis regards genotypic values as continuous or a discrete variable in two manners. First, genotypic values were categorized based on the absolute values depending on the trait, using two selection intensities (SI), moderate and extreme (<xref ref-type="supplementary-material" rid="SM3">Supplementary Figure S1</xref>). The moderate SI was created to mimic a balanced dataset regarding the selected and non-selected classes, while the extreme SI was created for an unbalanced dataset. For GY, the higher-yielding individuals were regarded as the best. For the extreme SI, about 10% of the higher-yielding genotypes were selected; and for the moderate SI, around 50% of the higher-yielding were selected for the moderate SI. Second, the selection for PH was based on a hypothetical ideotype. In this case, genotypic values between 1.95 and 2.05&#x2009;m (~60%; moderate SI) were selected; additionally, genotypic values between 1.90&#x2009;m and 2.10 (~90%; extreme SI) were regarded as selected. Notice that under the extreme SI, the 10% best were selected for GY, while for PH, the 10% out of type were eliminated. This approach was chosen because selecting the central 10% of hybrids for PH would result in only ~0.02&#x2009;m range within the &#x201C;selected&#x201D; class.</p>
<p>The categorized genotypes were used for classification tasks in the subsequent analysis. For this, the selected genotypes were attributed value 1, while the non-selected had value 0. Notice that the metrics used to evaluate the prediction models might have different meanings for GY and PH. For example, under extreme SI, the individuals regarded as selected compose a low proportion of the samples for GY, while for PH, they are the majority. Finally, the original continuous variables were used in regression tasks. In this case, the genotypic values were scaled using <inline-formula>
<mml:math id="M19">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mfenced>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>/</mml:mo>
<mml:mfenced>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, were <inline-formula>
<mml:math id="M20">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M21">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the minimum and maximum genotypic values, respectively.</p>
</sec>
</sec>
<sec id="sec7">
<title>Independent Variables</title>
<p>A graphical summary of the procedures explained hereon is presented in <xref rid="fig1" ref-type="fig">Figure 1</xref>.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption><p>Summarized general exemplification of the employed methodology. <bold>(A)</bold> Genomic data obtained from the pre-processing step; <bold>(B)</bold> Additive Genomic Relationship Matrix (GRM) obtained with VanRaden&#x2019;s method using the genomic matrix <bold>(A)</bold>; <bold>(C)</bold> <italic>DeepInsight</italic> pipeline: t-SNE decomposition of the genomic matrix <bold>(A)</bold> with original (dark gray) and rotated (light gray) marker coordinates; <bold>(D)</bold> Genomic images obtained with <italic>DeepInsight</italic>; one image is produced for each hybrid and all markers are represented in the image; <bold>(E)</bold> representation of the genotypes (0, 1, and 2, also white, light gray, and dark gray, respectively) in a genomic image; each pixel might comprise a single or multiple markers depending on the level of linkage disequilibrium in the population; <bold>(F)</bold> Genomic prediction methods: genomic BLUP (GBLUP), multilayer perceptron (MLP), and convolutional neural network (CNN); GBLUP and MLP used the GRM <bold>(B)</bold> as independent variable, while CNN used genomic images <bold>(D)</bold>; the neural networks were used as both regression and classification tasks; AutoKeras was used for hyperparameter search in MLP and CNN; <bold>(G)</bold> Simplified representation of the nested validation procedure.</p></caption>
<graphic xlink:href="fpls-13-845524-g001.tif"/>
</fig>
<sec id="sec8">
<title>Genomic Data Pre-processing</title>
<p>The parental inbred lines were genotyped with the Affymetrix<sup>&#x00AE;</sup> Axiom<sup>&#x00AE;</sup> Array of 614&#x2009;k SNPs (<xref ref-type="bibr" rid="ref44">Unterseer et al., 2014</xref>). The genomic data pre-processing was performed following the procedure presented in <xref ref-type="bibr" rid="ref16">Galli et al. (2020a)</xref> by: removing markers with low call rate (&#x003C;95%); removing markers with at least one heterozygote in the population; imputing missing (homozygous) data with the Synbreed-R package (<xref ref-type="bibr" rid="ref47">Wimmer et al., 2012</xref>); pruning with Plink v. 1.9 (<xref ref-type="bibr" rid="ref9">Chang et al., 2015</xref>) so the maximum linkage disequilibrium between markers is 0.9 to avoid high-level redundancy between marker information; building the hybrids synthetic genomic matrix; and, removing markers with minor allele frequency lower than 5%. After pre-processing, a total of 34,571 markers remained for further analysis. Principal component analysis (<xref ref-type="bibr" rid="ref25">Lyra et al., 2017</xref>; <xref ref-type="bibr" rid="ref35">Morosini et al., 2017</xref>), linkage disequilibrium decay (<xref ref-type="bibr" rid="ref35">Morosini et al., 2017</xref>), distribution of minor allele frequency, and heterozygosity (<xref ref-type="bibr" rid="ref16">Galli et al., 2020a</xref>) have been reported for this dataset.</p>
</sec>
<sec id="sec9">
<title>Genomic Relationship Matrix</title>
<p>The genomic information was transformed into two types of data for inclusion in prediction methods. The first type utilized was the additive GRM. We opted for <xref ref-type="bibr" rid="ref46">VanRaden's (2008)</xref> baseline method to determine the genomic relationship between genotypes. The relationship was obtained as <inline-formula>
<mml:math id="M22">
<mml:mrow>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:msup>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:msup>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M23">
<mml:mi mathvariant="normal">X</mml:mi>
</mml:math>
</inline-formula> is the scaled matrix of genotypic information and <inline-formula>
<mml:math id="M24">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> is the number of individuals. The GRM was obtained using the <italic>G.matrix</italic> function of the <italic>snpReady</italic> (<xref ref-type="bibr" rid="ref21">Granato et al., 2018</xref>) R library.</p>
</sec>
<sec id="sec10">
<title>Obtaining Images From Genomic Data</title>
<p>The second type of data transformation was performed by converting the structured genotype matrix by marker into images. This was achieved using the <italic>DeepInsight</italic> algorithm proposed by <xref ref-type="bibr" rid="ref39">Sharma et al. (2019)</xref>. In summary, the algorithm applies a similarity measuring/dimensionality reduction technique (e.g., t-SNE, kPCA) to obtain a Cartesian representation of the similarity between genomic markers in the population. At this step, one graph is produced, and each point represents a marker (<xref rid="fig1" ref-type="fig">Figure 1C</xref>, dark gray). In this context, if two markers are somehow related due to, e.g., linkage disequilibrium, they should have similar coordinates. Then, the algorithm finds the smallest rectangle containing all the points and applies a rotation to the graph, so the rectangle is vertically or horizontally oriented (<xref rid="fig1" ref-type="fig">Figure 1C</xref>, light gray). At this point, the graph is converted to an image, and the genomic marker information (e.g., 0, 1, or 2, according to the number of copies of the most frequent allele) is mapped to its corresponding position (<xref rid="fig1" ref-type="fig">Figure 1E</xref>). This procedure produces one image per hybrid (<xref rid="fig1" ref-type="fig">Figure 1D</xref>).</p>
<p>Using <italic>DeepInsight</italic>, images were generated for the 904 genotypes (<xref rid="fig1" ref-type="fig">Figure 1D</xref>). The genomic matrix mapped to images had 0, 1, and 2 coding (<xref rid="fig1" ref-type="fig">Figure 1E</xref>), commonly used to estimate additive effects of markers or additive GRMs in genomic prediction. The Cartesian plane marker coordinates were estimated using kPCA and t-SNE; no relevant difference was found on preliminary tests, and the latter was selected. The 120&#x2009;&#x00D7;&#x2009;120 pixels resolution presented adequate results regarding image size, given the number of markers and the available computational power. The <italic>DeepInsight</italic> algorithm is implemented in MATLAB and available at <ext-link xlink:href="http://www.alok-ai-lab.com" ext-link-type="uri">http://www.alok-ai-lab.com</ext-link>.</p>
</sec>
</sec>
<sec id="sec11">
<title>Genomic Prediction</title>
<sec id="sec12">
<title>Prediction Scenarios</title>
<p>The GP methods used were GBLUP (standard method), MLP (using the GRM), and CNN (using the genomic images obtained with <italic>DeepInsight</italic>; <xref rid="fig1" ref-type="fig">Figure 1F</xref>). GP was performed as regression and classification tasks, i.e., the dependent variable (GY or PH) was continuous or discrete, respectively. For the regression task, the evaluated scenarios were: (1) GBLUP; (2) MLP; and (3) CNN. For the classification task, the scenarios were: (1) MLP under moderate SI; (2) MLP under extreme SI; (3) CNN under moderate SI; and (4) CNN under extreme SI. Thus, these scenarios enabled estimating the effect of prediction methods (GBLUP vs. MLP vs. CNN) in the regression task; the impact of selection intensity (moderate vs. extreme) in the classification task; and the effect of and data type/prediction method [MLP (GRM) vs. CNN (genomic image)] on both regression and classification.</p>
</sec>
<sec id="sec13">
<title>GBLUP</title>
<p>GBLUP is a standard regression task and was performed using ASReml-R (<xref ref-type="bibr" rid="ref19">Gilmour et al., 2009</xref>) following the given linear model:</p>
<disp-formula id="E2">
<mml:math id="M25">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mstyle mathvariant="bold">
<mml:mn>1</mml:mn>
</mml:mstyle>
<mml:mi>&#x03BC;</mml:mi>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mstyle mathvariant="bold">
<mml:mn>3</mml:mn>
</mml:mstyle>
</mml:msub>
<mml:mi>h</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math id="M26">
<mml:mover accent="true">
<mml:mi>g</mml:mi>
<mml:mo>&#x005E;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> is the scaled vector of genotypic values of hybrids; <inline-formula>
<mml:math id="M27">
<mml:mrow>
<mml:mi>&#x03BC;</mml:mi>
<mml:mi mathvariant="normal">&#x2004;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x005C;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the overall mean; <inline-formula>
<mml:math id="M28">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>~</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mi>G</mml:mi>
<mml:msubsup>
<mml:mi>&#x03C3;</mml:mi>
<mml:mi>h</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the vector of genomic estimated breeding values, considering that <inline-formula>
<mml:math id="M29">
<mml:mi mathvariant="normal">G</mml:mi>
</mml:math>
</inline-formula> is the <xref ref-type="bibr" rid="ref46">VanRaden's (2008)</xref> additive relationship matrix; and <inline-formula>
<mml:math id="M30">
<mml:mi>e</mml:mi>
</mml:math>
</inline-formula>
<inline-formula>
<mml:math id="M31">
<mml:mrow>
<mml:mo>~</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mi>I</mml:mi>
<mml:msubsup>
<mml:mi>&#x03C3;</mml:mi>
<mml:mi>e</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is the residual vector; <inline-formula>
<mml:math id="M32">
<mml:mstyle mathvariant="bold">
<mml:mn>1</mml:mn>
</mml:mstyle>
</mml:math>
</inline-formula> vector of one for the intercept and <inline-formula>
<mml:math id="M33">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">Z</mml:mi>
<mml:mstyle mathvariant="bold">
<mml:mn>3</mml:mn>
</mml:mstyle>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the incidence matrix for genotypes.</p>
</sec>
<sec id="sec14">
<title>Neural Networks</title>
<p>We call the attention that this work is not focused on an in-depth explanation of neural networks as an algorithm despite the need for a basic understanding of neural networks. If the reader is not familiar with the subject, we encourage the reading of <xref ref-type="bibr" rid="ref20">Gonz&#x00E1;lez-Camacho et al. (2016)</xref> and <xref ref-type="bibr" rid="ref37">P&#x00E9;rez-Enciso and Zingaretti (2019)</xref> for a thorough comprehension of key concepts.</p>
<p>Neural networks were performed for regression and classification. The python AutoML system <italic>AutoKeras</italic> (<xref ref-type="bibr" rid="ref23">Jin et al., 2019</xref>) was used in this context. AutoML libraries perform neural architecture search with minor manual intervention and enable the automated finding of population-specific machine learning models. In this context, regression was implemented with the <italic>ImageRegressor</italic> and the <italic>StructuredDataRegressor</italic> functions to search suitable CNNs and MLPs, respectively. The loss function was the mean squared error (MSE;<inline-formula>
<mml:math id="M34">
<mml:mrow>
<mml:mspace width="thickmathspace"/>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) with <inline-formula>
<mml:math id="M35">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> computed as the difference between observed and predicted values and the metrics were: (a) the mean absolute error (MAE; <inline-formula>
<mml:math id="M36">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>), and (b) Pearson&#x2019;s product&#x2013;moment correlation (r; <inline-formula>
<mml:math id="M37">
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mfenced>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mfenced>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>/</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mfenced>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mfenced>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</inline-formula>), given that <inline-formula>
<mml:math id="M38">
<mml:mrow>
<mml:msub>
<mml:mi>e</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the residual for hybrid <inline-formula>
<mml:math id="M39">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M40">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the genotypic value of hybrid <inline-formula>
<mml:math id="M41">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M42">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted value of hybrid <inline-formula>
<mml:math id="M43">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math id="M44">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula> is the number of observations, <inline-formula>
<mml:math id="M45">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the mean of genotypic values, and <inline-formula>
<mml:math id="M46">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BC;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mi>g</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the mean of predicted values. Also, it is important to point out that these metrics were computed in training (inner-training), validation (inner-validation), and testing sets (outer-validation).</p>
<p>The classification was performed with the <italic>ImageClassifier</italic> and the <italic>StructuredDataClassifier</italic> functions to identify CNNs and MLPs. In this context, positives (<inline-formula>
<mml:math id="M47">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula>) are genotypes that would have been selected based on their genotypic value, and negatives (<inline-formula>
<mml:math id="M48">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula>) are non-selected genotypes. A loss function and several metrics were estimated based on the number of true positives (<inline-formula>
<mml:math id="M49">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), false positives (<inline-formula>
<mml:math id="M50">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), true negatives (<inline-formula>
<mml:math id="M51">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), and false negatives (<inline-formula>
<mml:math id="M52">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>). The loss function was the binary cross-entropy, and the metrics were true negative rate (TNR; <inline-formula>
<mml:math id="M53">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), precision [<inline-formula>
<mml:math id="M54">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>], recall (or true positive rate; TPR) [<inline-formula>
<mml:math id="M55">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>], F<sub>1</sub> score [<inline-formula>
<mml:math id="M56">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>], accuracy [<inline-formula>
<mml:math id="M57">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>t</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>], balanced accuracy [<inline-formula>
<mml:math id="M58">
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>], and area under the receiver operating characteristic curve (AUC). As the datasets were imbalanced, especially for the extreme selection intensity, the weights (<inline-formula>
<mml:math id="M59">
<mml:mi>w</mml:mi>
</mml:math>
</inline-formula>) of classes (selected and non-selected) were fed to the model given that <inline-formula>
<mml:math id="M60">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M61">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the frequency of class <inline-formula>
<mml:math id="M62">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula>.</p>
<p>For both classification and regression, the maximum number of models tried by AutoKeras was 50; the number of epochs was set to 150; the batch size was set to eight; seeds were utilized for reproducibility. Finally, the objective of the search was to identify the hyperparameters that minimized the validation loss.</p>
</sec>
<sec id="sec15">
<title>Classification/Regression Performance</title>
<p>A random sampling validation scheme assessed the prediction performance under each validation scenario (<xref rid="fig1" ref-type="fig">Figure 1G</xref>). For the neural networks (MLP and CNN), an adaptation of the validation was applied to steps 1 and 2, enabling the identification of the best set of hyperparameters for each replication within a scenario, a process called inner validation or also named model calibration, similar to the utilized by <xref ref-type="bibr" rid="ref31">Montesinos-L&#x00F3;pez et al. (2018a</xref>,<xref ref-type="bibr" rid="ref32">b)</xref>. The steps exclusively performed for neural networks are represented by lowercase letters.</p>
</sec>
<sec id="sec16">
<title>Model Calibration</title>
<p>The overall validation procedure was performed as follows:</p>
<list list-type="order">
<list-item>
<p>Allocation of randomly sampled genotypes to training (80%; TS) and validation sets (20%; <italic>VS</italic>):</p>
</list-item>
</list>
<list list-type="alpha-lower">
<list-item>
<p>Random assignment of samples from the training set into inner training (ITS; 80%) and inner validation sets (IVS; 20%);</p>
</list-item>
<list-item>
<p>Identification of the best set of hyperparameters with AutoKeras using the inner training and inner validation sets.</p>
</list-item>
</list>
<list>
<list-item>
<p>2. Fit model using all individuals from the inner sets (ITS and IVS);</p>
</list-item>
<list-item>
<p>3. Prediction of the outer validation set.</p>
</list-item>
<list-item>
<p>4. Estimation of comparison metrics; and</p>
</list-item>
<list-item>
<p>5. Repeat steps 1&#x2013;4 five times, considering equal set sampling between scenarios.</p>
</list-item>
</list>
<p>Comparisons between scenarios were made using the metrics estimated on the validation process. Values are presented as mean and standard deviation across the five replications.</p>
</sec>
</sec>
</sec>
<sec id="sec17" sec-type="results">
<title>Results</title>
<sec id="sec18">
<title>Phenotypic Analysis</title>
<p>According to the joint phenotypic analysis, the plot level repeatabilities were 0.23 for GY and 0.59 for PH, revealing the traits as lowly and moderately heritable, respectively. The LRT test (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.05) showed the effects of the environment, block within environment, genotype, and genotype (check) by environment to significantly affect both GY and PH. The BLUPs of GY averaged 6.79&#x2009;Mg&#x2009;ha<sup>&#x2212;1</sup> ranging from 4.90 to 8.36&#x2009;Mg&#x2009;ha<sup>&#x2212;1</sup>. For PH, the mean was 199.02&#x2009;cm, with values varying between 170.69 and 217.74&#x2009;cm (<xref ref-type="supplementary-material" rid="SM1">Supplementary Figure S1</xref>). More information on these genotypes can be found in <xref ref-type="bibr" rid="ref3">Alves et al. (2019)</xref>, <xref ref-type="bibr" rid="ref15">Fritsche-Neto et al. (2019)</xref>, and <xref ref-type="bibr" rid="ref16">Galli et al. (2020a)</xref>.</p>
</sec>
<sec id="sec19">
<title>Regression Performance</title>
<p>The regression metrics for the prediction of PH and GY using GBLUP, MLP, or CNN are presented in <xref rid="tab1" ref-type="table">Table 1</xref>. Considering each prediction scenario (e.g., PH with MLP), the metrics (MAE or MSE) were generally consistent across the inner training, inner validation, and outer validation sets. Therefore, scenario comparisons were performed, having the outer validation set as a reference. The values of MAE varied from 0.0635 to 0.0770 for PH and from 0.0797 to 0.0850 for GY. The loss function (MSE) varied from 0.0083 to 0.0109 for PH and from 0.0114 to 0.0128 for GY. The correlations varied from 0.68 to 0.75 for PH and from 0.53 to 0.59 for GY.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption><p>Regression metrics for dependent variables (DV) plant height (PH) and grain yield (GY) using genomic BLUP (GBLUP), Multilayer Perceptrons (MLP), and Convolutional Neural Networks (CNN).</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" colspan="2">Scenario</th>
<th align="center" valign="top" colspan="6">MAE</th>
<th align="center" valign="top" colspan="6">MSE (loss)</th>
<th align="center" valign="top" colspan="2">r</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">DV</td>
<td align="left" valign="middle">Method</td>
<td align="center" valign="middle" colspan="2">Inner training</td>
<td align="center" valign="middle" colspan="2">Inner validation</td>
<td align="center" valign="middle" colspan="2">Outer validation</td>
<td align="center" valign="middle" colspan="2">Inner training</td>
<td align="center" valign="middle" colspan="2">Inner validation</td>
<td align="center" valign="middle" colspan="2">Outer validation</td>
<td align="center" valign="middle" colspan="2">Outer validation</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">PH</td>
<td align="left" valign="bottom">GBLUP</td>
<td align="char" valign="bottom" char=".">0.0607</td>
<td align="char" valign="bottom" char=".">(0.0006)</td>
<td align="char" valign="bottom" char=".">-</td>
<td align="char" valign="bottom" char=".">-</td>
<td align="char" valign="bottom" char=".">0.0635</td>
<td align="char" valign="bottom" char=".">(0.0017)</td>
<td align="char" valign="bottom" char=".">0.0076</td>
<td align="char" valign="bottom" char=".">(0.0002)</td>
<td align="char" valign="bottom" char=".">-</td>
<td align="char" valign="bottom" char=".">-</td>
<td align="char" valign="bottom" char=".">0.0083</td>
<td align="char" valign="bottom" char=".">(0.0008)</td>
<td align="char" valign="bottom" char=".">0.75</td>
<td align="char" valign="bottom" char=".">(0.03)</td>
</tr>
<tr>
<td align="left" valign="middle">MLP</td>
<td align="char" valign="middle" char=".">0.0946</td>
<td align="char" valign="middle" char=".">(0.0552)</td>
<td align="char" valign="middle" char=".">0.0693</td>
<td align="char" valign="middle" char=".">(0.0087)</td>
<td align="char" valign="middle" char=".">0.0770</td>
<td align="char" valign="middle" char=".">(0.0164)</td>
<td align="char" valign="middle" char=".">0.0191</td>
<td align="char" valign="middle" char=".">(0.0227)</td>
<td align="char" valign="middle" char=".">0.0086</td>
<td align="char" valign="middle" char=".">(0.0022)</td>
<td align="char" valign="middle" char=".">0.0109</td>
<td align="char" valign="middle" char=".">(0.0039)</td>
<td align="char" valign="middle" char=".">0.70</td>
<td align="char" valign="middle" char=".">(0.06)</td>
</tr>
<tr>
<td align="left" valign="middle">CNN</td>
<td align="char" valign="middle" char=".">0.0693</td>
<td align="char" valign="middle" char=".">(0.0053)</td>
<td align="char" valign="middle" char=".">0.0642</td>
<td align="char" valign="middle" char=".">(0.0077)</td>
<td align="char" valign="middle" char=".">0.0750</td>
<td align="char" valign="middle" char=".">(0.0107)</td>
<td align="char" valign="middle" char=".">0.0087</td>
<td align="char" valign="middle" char=".">(0.0013)</td>
<td align="char" valign="middle" char=".">0.0077</td>
<td align="char" valign="middle" char=".">(0.0024)</td>
<td align="char" valign="middle" char=".">0.0103</td>
<td align="char" valign="middle" char=".">(0.0023)</td>
<td align="char" valign="middle" char=".">0.68</td>
<td align="char" valign="middle" char=".">(0.08)</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="3">GY</td>
<td align="left" valign="middle">GBLUP</td>
<td align="char" valign="middle" char=".">0.0736</td>
<td align="char" valign="middle" char=".">(0.0009)</td>
<td align="char" valign="middle" char=".">-</td>
<td align="char" valign="middle" char=".">-</td>
<td align="char" valign="middle" char=".">0.0836</td>
<td align="char" valign="middle" char=".">(0.0038)</td>
<td align="char" valign="middle" char=".">0.0098</td>
<td align="char" valign="middle" char=".">(0.0003)</td>
<td align="char" valign="middle" char=".">-</td>
<td align="char" valign="middle" char=".">-</td>
<td align="char" valign="middle" char=".">0.0128</td>
<td align="char" valign="middle" char=".">(0.0011)</td>
<td align="char" valign="middle" char=".">0.56</td>
<td align="char" valign="middle" char=".">(0.05)</td>
</tr>
<tr>
<td align="left" valign="middle">MLP</td>
<td align="char" valign="middle" char=".">0.0922</td>
<td align="char" valign="middle" char=".">(0.0142)</td>
<td align="char" valign="middle" char=".">0.0823</td>
<td align="char" valign="middle" char=".">(0.0043)</td>
<td align="char" valign="middle" char=".">0.0850</td>
<td align="char" valign="middle" char=".">(0.0048)</td>
<td align="char" valign="middle" char=".">0.0147</td>
<td align="char" valign="middle" char=".">(0.004)</td>
<td align="char" valign="middle" char=".">0.0114</td>
<td align="char" valign="middle" char=".">(0.0017)</td>
<td align="char" valign="middle" char=".">0.0125</td>
<td align="char" valign="middle" char=".">(0.0018)</td>
<td align="char" valign="middle" char=".">0.53</td>
<td align="char" valign="middle" char=".">(0.04)</td>
</tr>
<tr>
<td align="left" valign="middle">CNN</td>
<td align="char" valign="middle" char=".">0.0776</td>
<td align="char" valign="middle" char=".">(0.0117)</td>
<td align="char" valign="middle" char=".">0.0778</td>
<td align="char" valign="middle" char=".">(0.004)</td>
<td align="char" valign="middle" char=".">0.0797</td>
<td align="char" valign="middle" char=".">(0.0057)</td>
<td align="char" valign="middle" char=".">0.0106</td>
<td align="char" valign="middle" char=".">(0.0031)</td>
<td align="char" valign="middle" char=".">0.0102</td>
<td align="char" valign="middle" char=".">(0.0014)</td>
<td align="char" valign="middle" char=".">0.0114</td>
<td align="char" valign="middle" char=".">(0.0012)</td>
<td align="char" valign="middle" char=".">0.59</td>
<td align="char" valign="middle" char=".">(0.01)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Mean absolute error (MAE), mean squared error (MSE; loss function), and Pearson&#x2019;s product&#x2013;moment correlation (<italic>r</italic>) are presented for the inner training, inner validation, and outer validation sets. The values are the mean and standard deviations (in parenthesis) across five replications.</p>
</table-wrap-foot>
</table-wrap>
<p>Patterns arose from comparisons between the studied scenarios. Contrasting the GBLUP (standard model) with MLP and CNN, GBLUP yielded superior results across all metrics for PH. For GY, a similar pattern was observed when comparing GBLUP with MLP, except for MSE. However, CNN outperformed GBLUP for this trait concerning all metrics. Regarding MLP or CNN, the latter presented better results for both traits considering all estimated metrics, except for r in PH.</p>
</sec>
<sec id="sec20">
<title>Classification Performance</title>
<p>The classification metrics for the prediction of PH and GY using MLP or CNN are presented in <xref rid="tab2" ref-type="table">Table 2</xref> and <xref ref-type="supplementary-material" rid="SM1">Supplementary Table S1</xref>. Under each prediction scenario, the loss increased from inner training to inner validation to outer validation. Regarding the other metrics (with few exceptions), values were (in average across scenarios) greater in inner training, followed by inner validation and outer validation sets. In order to facilitate the evaluation, comparisons between scenarios were performed using the outer validation set as a reference.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption><p>Classification metrics for dependent variables (DV) plant height (PH) and grain yield (GY) using genomic BLUP (GBLUP), Multilayer Perceptrons (MLP), or Convolutional Neural Networks (CNN) under moderate and extreme selection intensities (SI).</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top" colspan="3">Scenario</th>
<th align="center" valign="top" colspan="2">TNR</th>
<th align="center" valign="top" colspan="2">Recall (TPR)</th>
<th align="center" valign="top" colspan="2">Precision</th>
<th align="center" valign="top" colspan="2">F1 score</th>
<th align="center" valign="top" colspan="2">Accuracy</th>
<th align="center" valign="top" colspan="2">Balanced accuracy</th>
<th align="center" valign="top" colspan="2">AUC</th>
<th align="center" valign="top" colspan="2">BC (loss)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">DV</td>
<td align="left" valign="middle">Method</td>
<td align="left" valign="middle">SI</td>
<td align="left" valign="middle" colspan="16">Inner training</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="4">PH</td>
<td align="left" valign="middle" rowspan="2">MLP</td>
<td align="left" valign="bottom">Extreme</td>
<td align="char" valign="bottom" char=".">0.91</td>
<td align="char" valign="bottom" char=".">(0.20)</td>
<td align="char" valign="bottom" char=".">0.95</td>
<td align="char" valign="bottom" char=".">(0.05)</td>
<td align="char" valign="bottom" char=".">0.99</td>
<td align="char" valign="bottom" char=".">(0.02)</td>
<td align="char" valign="bottom" char=".">0.97</td>
<td align="char" valign="bottom" char=".">(0.03)</td>
<td align="char" valign="bottom" char=".">0.95</td>
<td align="char" valign="bottom" char=".">(0.06)</td>
<td align="char" valign="bottom" char=".">0.93</td>
<td align="char" valign="bottom" char=".">(0.11)</td>
<td align="char" valign="bottom" char=".">0.94</td>
<td align="char" valign="bottom" char=".">(0.13)</td>
<td align="char" valign="bottom" char=".">0.0005</td>
<td align="char" valign="bottom" char=".">(0.0007)</td>
</tr>
<tr>
<td align="left" valign="middle">Moderate</td>
<td align="char" valign="middle" char=".">0.55</td>
<td align="char" valign="middle" char=".">(0.17)</td>
<td align="char" valign="middle" char=".">0.75</td>
<td align="char" valign="middle" char=".">(0.13)</td>
<td align="char" valign="middle" char=".">0.72</td>
<td align="char" valign="middle" char=".">(0.05)</td>
<td align="char" valign="middle" char=".">0.73</td>
<td align="char" valign="middle" char=".">(0.05)</td>
<td align="char" valign="middle" char=".">0.67</td>
<td align="char" valign="middle" char=".">(0.05)</td>
<td align="char" valign="middle" char=".">0.65</td>
<td align="char" valign="middle" char=".">(0.05)</td>
<td align="char" valign="middle" char=".">0.69</td>
<td align="char" valign="middle" char=".">(0.07)</td>
<td align="char" valign="middle" char=".">0.0014</td>
<td align="char" valign="middle" char=".">(0.0001)</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">CNN</td>
<td align="left" valign="middle">Extreme</td>
<td align="char" valign="middle" char=".">0.75</td>
<td align="char" valign="middle" char=".">(0.21)</td>
<td align="char" valign="middle" char=".">0.72</td>
<td align="char" valign="middle" char=".">(0.23)</td>
<td align="char" valign="middle" char=".">0.95</td>
<td align="char" valign="middle" char=".">(0.04)</td>
<td align="char" valign="middle" char=".">0.81</td>
<td align="char" valign="middle" char=".">(0.16)</td>
<td align="char" valign="middle" char=".">0.73</td>
<td align="char" valign="middle" char=".">(0.22)</td>
<td align="char" valign="middle" char=".">0.74</td>
<td align="char" valign="middle" char=".">(0.22)</td>
<td align="char" valign="middle" char=".">0.77</td>
<td align="char" valign="middle" char=".">(0.21)</td>
<td align="char" valign="middle" char=".">0.0022</td>
<td align="char" valign="middle" char=".">(0.0029)</td>
</tr>
<tr>
<td align="left" valign="middle">Moderate</td>
<td align="char" valign="middle" char=".">0.61</td>
<td align="char" valign="middle" char=".">(0.12)</td>
<td align="char" valign="middle" char=".">0.68</td>
<td align="char" valign="middle" char=".">(0.08)</td>
<td align="char" valign="middle" char=".">0.72</td>
<td align="char" valign="middle" char=".">(0.07)</td>
<td align="char" valign="middle" char=".">0.70</td>
<td align="char" valign="middle" char=".">(0.07)</td>
<td align="char" valign="middle" char=".">0.65</td>
<td align="char" valign="middle" char=".">(0.09)</td>
<td align="char" valign="middle" char=".">0.64</td>
<td align="char" valign="middle" char=".">(0.09)</td>
<td align="char" valign="middle" char=".">0.68</td>
<td align="char" valign="middle" char=".">(0.12)</td>
<td align="char" valign="middle" char=".">0.0014</td>
<td align="char" valign="middle" char=".">(0.0002)</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="4">GY</td>
<td align="left" valign="middle" rowspan="2">MLP</td>
<td align="left" valign="middle">Extreme</td>
<td align="char" valign="middle" char=".">0.76</td>
<td align="char" valign="middle" char=".">(0.26)</td>
<td align="char" valign="middle" char=".">0.91</td>
<td align="char" valign="middle" char=".">(0.12)</td>
<td align="char" valign="middle" char=".">0.46</td>
<td align="char" valign="middle" char=".">(0.27)</td>
<td align="char" valign="middle" char=".">0.58</td>
<td align="char" valign="middle" char=".">(0.28)</td>
<td align="char" valign="middle" char=".">0.78</td>
<td align="char" valign="middle" char=".">(0.24)</td>
<td align="char" valign="middle" char=".">0.84</td>
<td align="char" valign="middle" char=".">(0.17)</td>
<td align="char" valign="middle" char=".">0.86</td>
<td align="char" valign="middle" char=".">(0.18)</td>
<td align="char" valign="middle" char=".">0.0009</td>
<td align="char" valign="middle" char=".">(0.0007)</td>
</tr>
<tr>
<td align="left" valign="middle">Moderate</td>
<td align="char" valign="middle" char=".">0.72</td>
<td align="char" valign="middle" char=".">(0.04)</td>
<td align="char" valign="middle" char=".">0.74</td>
<td align="char" valign="middle" char=".">(0.01)</td>
<td align="char" valign="middle" char=".">0.73</td>
<td align="char" valign="middle" char=".">(0.03)</td>
<td align="char" valign="middle" char=".">0.73</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.73</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.73</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.80</td>
<td align="char" valign="middle" char=".">(0.03)</td>
<td align="char" valign="middle" char=".">0.0012</td>
<td align="char" valign="middle" char=".">(0.0001)</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">CNN</td>
<td align="left" valign="middle">Extreme</td>
<td align="char" valign="middle" char=".">0.84</td>
<td align="char" valign="middle" char=".">(0.13)</td>
<td align="char" valign="middle" char=".">0.88</td>
<td align="char" valign="middle" char=".">(0.12)</td>
<td align="char" valign="middle" char=".">0.51</td>
<td align="char" valign="middle" char=".">(0.32)</td>
<td align="char" valign="middle" char=".">0.61</td>
<td align="char" valign="middle" char=".">(0.28)</td>
<td align="char" valign="middle" char=".">0.85</td>
<td align="char" valign="middle" char=".">(0.13)</td>
<td align="char" valign="middle" char=".">0.86</td>
<td align="char" valign="middle" char=".">(0.12)</td>
<td align="char" valign="middle" char=".">0.91</td>
<td align="char" valign="middle" char=".">(0.10)</td>
<td align="char" valign="middle" char=".">0.0007</td>
<td align="char" valign="middle" char=".">(0.0006)</td>
</tr>
<tr>
<td align="left" valign="middle">Moderate</td>
<td align="char" valign="middle" char=".">0.74</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.76</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.75</td>
<td align="char" valign="middle" char=".">(0.01)</td>
<td align="char" valign="middle" char=".">0.75</td>
<td align="char" valign="middle" char=".">(0.01)</td>
<td align="char" valign="middle" char=".">0.75</td>
<td align="char" valign="middle" char=".">(0.01)</td>
<td align="char" valign="middle" char=".">0.75</td>
<td align="char" valign="middle" char=".">(0.01)</td>
<td align="char" valign="middle" char=".">0.83</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.0011</td>
<td align="char" valign="middle" char=".">(0.0001)</td>
</tr>
<tr>
<td align="left" valign="middle">DV</td>
<td align="left" valign="middle">Method</td>
<td align="left" valign="middle">SI</td>
<td align="left" valign="middle" colspan="16">Inner validation</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="4">PH</td>
<td align="left" valign="middle" rowspan="2">MLP</td>
<td align="left" valign="middle">Extreme</td>
<td align="char" valign="middle" char=".">0.27</td>
<td align="char" valign="middle" char=".">(0.18)</td>
<td align="char" valign="middle" char=".">0.94</td>
<td align="char" valign="middle" char=".">(0.04)</td>
<td align="char" valign="middle" char=".">0.92</td>
<td align="char" valign="middle" char=".">(0.03)</td>
<td align="char" valign="middle" char=".">0.93</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.88</td>
<td align="char" valign="middle" char=".">(0.03)</td>
<td align="char" valign="middle" char=".">0.61</td>
<td align="char" valign="middle" char=".">(0.08)</td>
<td align="char" valign="middle" char=".">0.68</td>
<td align="char" valign="middle" char=".">(0.16)</td>
<td align="char" valign="middle" char=".">0.3454</td>
<td align="char" valign="middle" char=".">(0.0951)</td>
</tr>
<tr>
<td align="left" valign="middle">Moderate</td>
<td align="char" valign="middle" char=".">0.50</td>
<td align="char" valign="middle" char=".">(0.16)</td>
<td align="char" valign="middle" char=".">0.76</td>
<td align="char" valign="middle" char=".">(0.12)</td>
<td align="char" valign="middle" char=".">0.68</td>
<td align="char" valign="middle" char=".">(0.06)</td>
<td align="char" valign="middle" char=".">0.71</td>
<td align="char" valign="middle" char=".">(0.03)</td>
<td align="char" valign="middle" char=".">0.64</td>
<td align="char" valign="middle" char=".">(0.04)</td>
<td align="char" valign="middle" char=".">0.63</td>
<td align="char" valign="middle" char=".">(0.04)</td>
<td align="char" valign="middle" char=".">0.70</td>
<td align="char" valign="middle" char=".">(0.06)</td>
<td align="char" valign="middle" char=".">0.6003</td>
<td align="char" valign="middle" char=".">(0.0399)</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="2">CNN</td>
<td align="left" valign="middle">Extreme</td>
<td align="char" valign="middle" char=".">0.12</td>
<td align="char" valign="middle" char=".">(0.16)</td>
<td align="char" valign="middle" char=".">0.99</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.91</td>
<td align="char" valign="middle" char=".">(0.03)</td>
<td align="char" valign="middle" char=".">0.95</td>
<td align="char" valign="middle" char=".">(0.02)</td>
<td align="char" valign="middle" char=".">0.90</td>
<td align="char" valign="middle" char=".">(0.03)</td>
<td align="char" valign="middle" char=".">0.55</td>
<td align="char" valign="middle" char=".">(0.07)</td>
<td align="char" valign="middle" char=".">0.73</td>
<td align="char" valign="middle" char=".">(0.14)</td>
<td align="char" valign="middle" char=".">0.2863</td>
<td align="char" valign="middle" char=".">(0.0640)</td>
</tr>
<tr>
<td align="left" valign="middle">Moderate</td>
<td align="char" valign="middle" char=".">0.47</td>
<td align="char" valign="middle" char=".">(0.13)</td>
<td align="char" valign="middle" char=".">0.80</td>
<td align="char" valign="middle" char=".">(0.10)</td>
<td align="char" valign="middle" char=".">0.67</td>
<td align="char" valign="middle" char=".">(0.06)</td>
<td align="char" valign="middle" char=".">0.73</td>
<td align="char" valign="middle" char=".">(0.07)</td>
<td align="char" valign="middle" char=".">0.66</td>
<td align="char" valign="middle" char=".">(0.08)</td>
<td align="char" valign="middle" char=".">0.63</td>
<td align="char" valign="middle" char=".">(0.08)</td>
<td align="char" valign="middle" char=".">0.70</td>
<td align="char" valign="middle" char=".">(0.05)</td>
<td align="char" valign="middle" char=".">0.6135</td>
<td align="char" valign="middle" char=".">(0.0402)</td>
</tr>
<tr>
<td align="left" valign="middle" rowspan="4">GY</td>
<td align="left" valign="middle" rowspan="2">MLP</td>
<td align="left" valign="middle">Extreme</td>
<td align="char" valign="middle" char=".">0.94</td>
<td align="char" valign="middle" char=".">(0.05)</td>
<td align="char" valign="middle" char=".">0.38</td>
<td align="char" valign="middle" char=".">(0.23)</td>
<td align="char" valign="middle" char=".">0.37</td>
<td align="char" valign="middle" char=".">(0.22)</td>
<td align="char" valign="middle" char=".">0.37</td>
<td align="char" valign="middle" char=".">(0.21)</td>
<td align="char" valign="middle" char=".">0.88</td>
<td align="char" valign="middle" char=".">(0.04)</td>
<td align="char" valign="middle" char=".">0.66</td>
<td align="char" valign="middle" char=".">(0.10)</td>
<td align="char" valign="middle" char=".">0.86</td>
<td align="char" valign="middle" char=".">(0.08)</td>
<td align="char" valign="middle" char=".">0.2621</td>
<td align="char" valign="middle" char=".">(0.0755)</td>
</tr>
<tr>
<td align="left" valign="middle">Moderate</td>
<td align="char" valign="middle" char=".">0.75</td>
<td align="char" valign="middle" char=".">(0.06)</td>
<td align="char" valign="middle" char=".">0.71</td>
<td align="char" valign="middle" char=".">(0.06)</td>
<td align="char" valign="middle" char=".">0.74</td>
<td align="char" valign="middle" char=".">(0.05)</td>
<td align="char" valign="top" char=".">0.72</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">0.73</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.73</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.80</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.5427</td>
<td align="char" valign="top" char=".">(0.0159)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">CNN</td>
<td align="left" valign="top">Extreme</td>
<td align="char" valign="top" char=".">0.97</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.31</td>
<td align="char" valign="top" char=".">(0.20)</td>
<td align="char" valign="top" char=".">0.44</td>
<td align="char" valign="top" char=".">(0.27)</td>
<td align="char" valign="top" char=".">0.35</td>
<td align="char" valign="top" char=".">(0.21)</td>
<td align="char" valign="top" char=".">0.90</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.64</td>
<td align="char" valign="top" char=".">(0.09)</td>
<td align="char" valign="top" char=".">0.84</td>
<td align="char" valign="top" char=".">(0.07)</td>
<td align="char" valign="top" char=".">0.2651</td>
<td align="char" valign="top" char=".">(0.0508)</td>
</tr>
<tr>
<td align="left" valign="top">Moderate</td>
<td align="char" valign="top" char=".">0.71</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">0.72</td>
<td align="char" valign="top" char=".">(0.09)</td>
<td align="char" valign="top" char=".">0.70</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.71</td>
<td align="char" valign="top" char=".">(0.05)</td>
<td align="char" valign="top" char=".">0.72</td>
<td align="char" valign="top" char=".">(0.03)</td>
<td align="char" valign="top" char=".">0.72</td>
<td align="char" valign="top" char=".">(0.03)</td>
<td align="char" valign="top" char=".">0.79</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.5486</td>
<td align="char" valign="top" char=".">(0.0188)</td>
</tr>
<tr>
<td align="left" valign="top">DV</td>
<td align="left" valign="top">Method</td>
<td align="left" valign="top">SI</td>
<td align="left" valign="top" colspan="16">Outer validation</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="4">PH</td>
<td align="left" valign="top" rowspan="2">MLP</td>
<td align="left" valign="top">Extreme</td>
<td align="char" valign="top" char=".">0.26</td>
<td align="char" valign="top" char=".">(0.21)</td>
<td align="char" valign="top" char=".">0.96</td>
<td align="char" valign="top" char=".">(0.03)</td>
<td align="char" valign="top" char=".">0.92</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.93</td>
<td align="char" valign="top" char=".">(0.01)</td>
<td align="char" valign="top" char=".">0.88</td>
<td align="char" valign="top" char=".">(0.01)</td>
<td align="char" valign="top" char=".">0.61</td>
<td align="char" valign="top" char=".">(0.09)</td>
<td align="char" valign="top" char=".">0.66</td>
<td align="char" valign="top" char=".">(0.11)</td>
<td align="char" valign="top" char=".">0.7612</td>
<td align="char" valign="top" char=".">(0.0904)</td>
</tr>
<tr>
<td align="left" valign="top">Moderate</td>
<td align="char" valign="top" char=".">0.65</td>
<td align="char" valign="top" char=".">(0.18)</td>
<td align="char" valign="top" char=".">0.57</td>
<td align="char" valign="top" char=".">(0.32)</td>
<td align="char" valign="top" char=".">0.71</td>
<td align="char" valign="top" char=".">(0.05)</td>
<td align="char" valign="top" char=".">0.57</td>
<td align="char" valign="top" char=".">(0.28)</td>
<td align="char" valign="top" char=".">0.59</td>
<td align="char" valign="top" char=".">(0.14)</td>
<td align="char" valign="top" char=".">0.61</td>
<td align="char" valign="top" char=".">(0.07)</td>
<td align="char" valign="top" char=".">0.67</td>
<td align="char" valign="top" char=".">(0.07)</td>
<td align="char" valign="top" char=".">1.1120</td>
<td align="char" valign="top" char=".">(0.4637)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">CNN</td>
<td align="left" valign="top">Extreme</td>
<td align="char" valign="top" char=".">0.32</td>
<td align="char" valign="top" char=".">(0.27)</td>
<td align="char" valign="top" char=".">0.79</td>
<td align="char" valign="top" char=".">(0.35)</td>
<td align="char" valign="top" char=".">0.90</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.79</td>
<td align="char" valign="top" char=".">(0.29)</td>
<td align="char" valign="top" char=".">0.73</td>
<td align="char" valign="top" char=".">(0.28)</td>
<td align="char" valign="top" char=".">0.55</td>
<td align="char" valign="top" char=".">(0.05)</td>
<td align="char" valign="top" char=".">0.63</td>
<td align="char" valign="top" char=".">(0.08)</td>
<td align="char" valign="top" char=".">1.0639</td>
<td align="char" valign="top" char=".">(0.4752)</td>
</tr>
<tr>
<td align="left" valign="top">Moderate</td>
<td align="char" valign="top" char=".">0.49</td>
<td align="char" valign="top" char=".">(0.15)</td>
<td align="char" valign="top" char=".">0.63</td>
<td align="char" valign="top" char=".">(0.22)</td>
<td align="char" valign="top" char=".">0.65</td>
<td align="char" valign="top" char=".">(0.06)</td>
<td align="char" valign="top" char=".">0.63</td>
<td align="char" valign="top" char=".">(0.16)</td>
<td align="char" valign="top" char=".">0.58</td>
<td align="char" valign="top" char=".">(0.07)</td>
<td align="char" valign="top" char=".">0.56</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">0.61</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">1.0838</td>
<td align="char" valign="top" char=".">(0.4774)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="4">GY</td>
<td align="left" valign="top" rowspan="2">MLP</td>
<td align="left" valign="top">Extreme</td>
<td align="char" valign="top" char=".">0.95</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">0.40</td>
<td align="char" valign="top" char=".">(0.23)</td>
<td align="char" valign="top" char=".">0.35</td>
<td align="char" valign="top" char=".">(0.22)</td>
<td align="char" valign="top" char=".">0.37</td>
<td align="char" valign="top" char=".">(0.21)</td>
<td align="char" valign="top" char=".">0.90</td>
<td align="char" valign="top" char=".">(0.03)</td>
<td align="char" valign="top" char=".">0.67</td>
<td align="char" valign="top" char=".">(0.10)</td>
<td align="char" valign="top" char=".">0.71</td>
<td align="char" valign="top" char=".">(0.13)</td>
<td align="char" valign="top" char=".">0.6838</td>
<td align="char" valign="top" char=".">(0.3174)</td>
</tr>
<tr>
<td align="left" valign="top">Moderate</td>
<td align="char" valign="top" char=".">0.66</td>
<td align="char" valign="top" char=".">(0.10)</td>
<td align="char" valign="top" char=".">0.67</td>
<td align="char" valign="top" char=".">(0.05)</td>
<td align="char" valign="top" char=".">0.67</td>
<td align="char" valign="top" char=".">(0.07)</td>
<td align="char" valign="top" char=".">0.67</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">0.66</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">0.67</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">0.72</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">2.2193</td>
<td align="char" valign="top" char=".">(2.5988)</td>
</tr>
<tr>
<td align="left" valign="top" rowspan="2">CNN</td>
<td align="left" valign="top">Extreme</td>
<td align="char" valign="top" char=".">0.97</td>
<td align="char" valign="top" char=".">(0.01)</td>
<td align="char" valign="top" char=".">0.17</td>
<td align="char" valign="top" char=".">(0.13)</td>
<td align="char" valign="top" char=".">0.37</td>
<td align="char" valign="top" char=".">(0.19)</td>
<td align="char" valign="top" char=".">0.22</td>
<td align="char" valign="top" char=".">(0.13)</td>
<td align="char" valign="top" char=".">0.90</td>
<td align="char" valign="top" char=".">(0.01)</td>
<td align="char" valign="top" char=".">0.57</td>
<td align="char" valign="top" char=".">(0.06)</td>
<td align="char" valign="top" char=".">0.72</td>
<td align="char" valign="top" char=".">(0.08)</td>
<td align="char" valign="top" char=".">0.7877</td>
<td align="char" valign="top" char=".">(0.4112)</td>
</tr>
<tr>
<td align="left" valign="top">Moderate</td>
<td align="char" valign="top" char=".">0.72</td>
<td align="char" valign="top" char=".">(0.07)</td>
<td align="char" valign="top" char=".">0.64</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">0.70</td>
<td align="char" valign="top" char=".">(0.08)</td>
<td align="char" valign="top" char=".">0.67</td>
<td align="char" valign="top" char=".">(0.03)</td>
<td align="char" valign="top" char=".">0.68</td>
<td align="char" valign="top" char=".">(0.02)</td>
<td align="char" valign="top" char=".">0.68</td>
<td align="char" valign="top" char=".">(0.03)</td>
<td align="char" valign="top" char=".">0.74</td>
<td align="char" valign="top" char=".">(0.04)</td>
<td align="char" valign="top" char=".">1.6882</td>
<td align="char" valign="top" char=".">(0.7615)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>True negative rate (TNR), recall (or true positive rate; TPR), precision, F1 score, accuracy, balanced accuracy, AUC, and binary cross-entropy (BC; loss function) are presented for the inner training, inner validation, and outer validation sets. The values are the mean and standard deviations (in parenthesis) across five replications.</p>
</table-wrap-foot>
</table-wrap>
<p>The observed values of metrics varied depending on the prediction scenario. The TNR varied from 0.26 to 0.65 for PH and 0.66 to 0.97 for GY; the recall ranged from 0.54 to 0.96 for PH and from 0.17 to 0.67 for GY; the precision presented values from 0.65 to 0.92 for PH and from 0.35 to 0.70 for GY; the F1score showed results from 0.57 to 0.93 for PH and from 0.22 to 0.67 for GY; the accuracy varied largely presenting values from 0.52 to 0.88 for PH and 0.66 to 0.90 for GY; the variation of balanced accuracy varied from 0.55 to 0.61 for PH and from 0.57 to 0.68 for GY; at last, the AUC ranged from 0.61 to 0.67 for PH and from 0.71 to 0.74 for GY.</p>
<p>The effect of selection intensity (moderate and extreme) presented tendencies to the estimated metrics. TNR, precision, recall, and F1 score were higher at extreme selection intensity for PH. For GY, the opposite was observed. The accuracy was higher at extreme selection intensity for both traits. The balanced accuracies using GRM were equal for both selection intensities for PH and GY. However, when CNN was used, this metric was lower at extreme selection intensity for both traits.</p>
<p>At last, regarding the AUC, moderate-intensity presented better values for both traits. Regarding the effect of the prediction method, for predicting PH, using MLP showed better values of precision, accuracy, balanced accuracy, and AUC. However, for recall and F1 score, this was only observed at extreme intensity. For GY, using MLP generally presented better results at extreme selection intensity, while CNN was superior at moderate selection intensity. The exceptions were precision, where image-based models were better for both intensities, and recall, which presented the opposite behavior.</p>
</sec>
<sec id="sec21">
<title>Automated Machine Learning Model Tuning</title>
<p>The neural network structures that minimized the loss function for each replication under each scenario are presented on <xref ref-type="supplementary-material" rid="SM2">Supplementary File S1</xref>. The classification scenarios were constitutionally composed of an input layer as the first, a dense layer as the second-last summarizing all the neurons of the previous layer, and an activation layer with the sigmoid function to generate the output probabilities. Similarly, the regression scenarios had an input layer as the first and a dense layer as the last to summarize all neurons to one output. Nevertheless, the network structures were generally different, with few exceptions. Among these coincidences, seven out of nine were of the same task type (regression or classification), four were of the same trait (PH or GY), and three were of the same selection intensity (extreme or moderate). However, the number of parameters varied greatly (from 24,533 to 23,589,764), typically higher when images were used.</p>
<p>Dealing with GRM or images requires networks with specific internal layers. The scenarios with MLPs presented a varying number of dense layers (1 to 4); normalization layers (0 to 4; present in about half of the networks); ReLU activation function (positioned after dense layers except the last one); and dropout (0 to 4; present in about 2/3 of networks). The CNNs were composed of 2-dimensional convolutions (1 to 4 in classifications and 2 to 6 in regressions; present in all networks); normalization layers (0 or 1; present in about 2/3 of the networks); 2-dimensional max/global max or average pooling (0 to 3; present in about 2/3 of networks); dropout (0 to 3; present in about 2/3 of networks); image processing filters (resize, random flip, contrast, rotation, translation, and concatenation; 0 to 4; present in about half of the networks; being more common for PH). Also, ResNet50 and Xception networks appeared within 1/3 of the classification networks (more common for PH).</p>
<p>Finally, the preferred optimizer was Adam; Adadelta and SGD also appeared in a limited number of cases. The most common learning rate was 0.001, followed by 0.01, 0.00001, 0.0001, and 0.1. The dropout regularization had values of 0.5 (most common) and 0.25.</p>
</sec>
</sec>
<sec id="sec22" sec-type="discussions">
<title>Discussion</title>
<sec id="sec23">
<title>Regression Analysis &#x2013; The Standard</title>
<p>Benchmark studies suggest the inconsistent performance of neural networks compared to standard GP methods, which depends on a series of factors. We contrasted our findings to reference studies and explored how these factors might have affected the results in this context. Concerning the regression analysis, the GBLUP method outperformed MLP for both traits. One of the factors reported determining the best methodology is how modeling is performed. GP was carried out as a two-stage analysis. Hence, the genotypic value of hybrids across environments was obtained before prediction. Accordingly, the environmental source of variation was absent and could not be captured by the ML methods. For instance, it has been extensively shown that linear models (e.g., GBLUP or BMTME) tend to be outperformed by MLP in a multi-environmental joint analysis if the genotype by environment factor is not modeled for the prediction of PH and GY in maize. This holds under both single (<xref ref-type="bibr" rid="ref32">Montesinos-L&#x00F3;pez et al., 2018b</xref>) and multi-trait (<xref ref-type="bibr" rid="ref31">Montesinos-L&#x00F3;pez et al., 2018a</xref>) modeling contexts. Accordingly, MLP was outperformed by GBLUP for both traits in our study, supporting the suggested effect of modeling to the comparative outcome for the studied GP methods.</p>
<p>The use of CNN presented better results than GBLUP and MLP for predicting GY. This contrasts with the findings of <xref ref-type="bibr" rid="ref4">Azodi et al. (2019)</xref>, who suggested that Ridge regression BLUP, a GBLUP-equivalent method, outperforms both MLP and CNN for predicting several traits on numerous crops, including PH and GY in maize inbred lines. In this case, CNN was the poorest performing method for both traits. The inconsistency between the results of these studies regarding the performance of the CNNs for GY could be attributed to the restrictive search space for hyperparameters given the computational requirements for the analysis of an astonishing amount of studied traits and species by <xref ref-type="bibr" rid="ref4">Azodi et al. (2019)</xref>; we tailored ML models to each scenario within each trait, which is known to improve NN performance (<xref ref-type="bibr" rid="ref31">Montesinos-L&#x00F3;pez et al., 2018a</xref>). Another factor that might have led to this discrepancy was the use of pre-processed genomic information (genomic images) in our CNNs. They opted for using the raw genomic matrix. At last, inbred lines were used in their work, while hybrids were used in ours; studies suggested that CNNs tend to have better performance (than linear methods) when strong nonlinear (e.g., dominance) effects are present (<xref ref-type="bibr" rid="ref6">Bellot et al., 2018</xref>; <xref ref-type="bibr" rid="ref1">Abdollahi-Arpanahi et al., 2020</xref>); which is the case of GY in population we studied (<xref ref-type="bibr" rid="ref3">Alves et al., 2019</xref>).</p>
<p>Regarding the underperformance of NN methods at predicting PH, tangible reasons could be pointed out. It has been hypothesized that the occurrence of extreme allelic frequencies (e.g., only two genotypes are present for a given <italic>locus</italic>) favors linear models by enabling the capture dominance and epistatic variance (<xref ref-type="bibr" rid="ref4">Azodi et al., 2019</xref>); however, this does not hold for this dataset (<xref ref-type="bibr" rid="ref16">Galli et al., 2020a</xref>). Also, PH is predominantly governed by additive allelic interactions (<xref ref-type="bibr" rid="ref3">Alves et al., 2019</xref>), which enables linear models to capture a considerable proportion of the genotypic variance; nevertheless, regardless of the nature of the effects governing the traits under study, ML should always be (at least) as good as linear models given their ability to model linear relationships (<xref ref-type="bibr" rid="ref4">Azodi et al., 2019</xref>), which was not the case. At last, a cause could be the number of training samples, which might not have been enough for modeling linear and nonlinear interactions between markers by the NN. This is a problem of common occurrence in plant breeding given the usually low number of samples, a large number of markers, and heterogeneity of data (<xref ref-type="bibr" rid="ref1">Abdollahi-Arpanahi et al., 2020</xref>; <xref ref-type="bibr" rid="ref38">Pook et al., 2020</xref>).</p>
<p>Overall, CNN presented better results compared to MLP. This advantage might have been due to the processing of the genomic matrix into the additive GRM, in the case of MLP. In this case, only the linear relationship between genotypes was modeled, which might compromise the potential of MLP to identify nonlinear effects. The genomic matrix could be used for further studies at the expense of computational time to overcome this issue.</p>
</sec>
<sec id="sec24">
<title>Classification Analysis &#x2013; The Alternative</title>
<p>Given their complex genetic nature, most plant traits present continuous phenotypes. Moreover, traits that were previously discretized by means of measurement ease, such as resistance to biotic stresses, have had their continuous nature better explored by high-throughput phenotyping (<xref ref-type="bibr" rid="ref17">Galli et al., 2020b</xref>). Therefore, regression tasks are an adequate fit for genetic analysis, including GP. Nevertheless, plant breeding is globally a classification problem in which genotypes are assigned classes (<xref ref-type="bibr" rid="ref36">Ornella et al., 2014</xref>), usually selected and non-selected. Hence, we elaborate on this problem, unifying ranking and selection by using classifying predictors. These prediction machines were evaluated using metrics that assess the model&#x2019;s ability to distinguish which genotypes should be selected.</p>
<p>A critical step on classification tasks is the discretization of the continuous variable; when applicable. Discretizing traits has its inherent degree of subjectivity, regarding, e.g., the number of classes and which threshold values are used to classify the data. Accordingly, these choices have been reported to influence the performance of prediction models (<xref ref-type="bibr" rid="ref36">Ornella et al., 2014</xref>; <xref ref-type="bibr" rid="ref20">Gonz&#x00E1;lez-Camacho et al., 2016</xref>). Furthermore, a greater level of subjectivity is introduced when genotype classification as true positives or negatives before the prediction is based on the empirical distribution rather than the absolute value of the trait. Predicting genotypes from a related population, classification would not be tied to the percentiles of the distribution on the training population but to the genotypic values and genetic variants under each class and the genetic similarities across populations. In this context, the algorithm might be targeting, e.g., plants with a height between 1.95 and 2.05&#x2009;m, but not the 10% or 50% best yielding hybrids since the distribution of genotypic values of a new population is likely to differ from the training population.</p>
<p>The classification problem was approached considering two scenarios: one highly imbalanced, where the size of the classes differed substantially (extreme SI), and one nearly balanced, where each class contained about half of the individuals (moderate SI). Both scenarios are plausible and of common occurrence in plant breeding, depending on the program stage. Nevertheless, imbalanced datasets should be evaluated with further cautiousness (<xref ref-type="bibr" rid="ref12">Fern&#x00E1;ndez et al., 2011</xref>). TNR, precision, recall, F1 score, accuracy, and AUC are examples of metrics sensitive to class imbalance, meaning their results might not be directly interpretable for comparing predictions with differing selection intensities. This is also evidenced by the discrepancy between the accuracy and the balanced accuracy at extreme selection intensity. The selection intensities presented little influence over the balanced accuracy for the same trait and independent variable, except for GY when images were used.</p>
<p>The balanced accuracy is calculated by averaging the proportion of correct predictions in each class, meaning that the label (selected or non-selected) is not relevant. This metric varied from 0.55 to 0.61 for PH and 0.57 to 0.68 for GY. These results are inconsistent with the regression analysis, which showed higher predictability for PH according to all metrics, following the higher heritability of this trait. We postulate that this is associated with the region of the empirical density of genotypic values from which genotypes were regarded as &#x201C;selected&#x201D;. For PH, the distinction between the best and the worst individuals was non-directional, which might have difficulted the distinction between which hybrids should or not be selected by the models. For GY, as the selection is directional, this was not an issue. Overall, balanced accuracies were closer to 0.5 (random guess) than to 1 (all correct) for both traits, meaning that further improvements are required. Nevertheless, our results suggest the possibility of non-directional selection, as for PH, which is highly relevant for breeding programs.</p>
<p>Unlike the regression task, where the use of CNN usually presented the best results between machine learning methods, there was considerable inconsistency regarding the superiority of MLP or CNN in the classification task. The comparative performance of the neural network methodologies seemed highly conditioned to trait and selection intensity. Generally, MLP presented the best results for PH, while for GY, the best method heavily depended on the selection intensity. Therefore, it is reasonable to assume that the discretization process of PH and GY impacted the performance of CNN more than that of MLP; but further investigation is warranted. GP prediction regression tasks with machine learning models are already common, but studies comparing methods for predicting discretized variables are still limited.</p>
</sec>
<sec id="sec25">
<title>Choosing Machine Learning Architectures</title>
<p>The choice of the neural network hyperparameters has been a critical step for NN-based GP by extensive benchmarking (<xref ref-type="bibr" rid="ref4">Azodi et al., 2019</xref>). Ergo, network search for a given task and dataset has been applied in recent ML-based GP studies (<xref ref-type="bibr" rid="ref32">Montesinos-L&#x00F3;pez et al., 2018b</xref>; <xref ref-type="bibr" rid="ref4">Azodi et al., 2019</xref>; <xref ref-type="bibr" rid="ref1">Abdollahi-Arpanahi et al., 2020</xref>). However, model tuning has been primarily performed using na&#x00EF;ve approaches such as random (values sampled from distribution) or grid (discrete values) search, which may limit the number of hyperparameter combinations based on a set of user-defined <italic>a priori</italic> information (<xref ref-type="bibr" rid="ref23">Jin et al., 2019</xref>). Due to recent advancements in computer science and technology, less restrictive, free, and easy-to-use hyperparameter search algorithms have been made available. Hence, we used Auto-Keras, an AutoML search algorithm with Bayesian optimization to identify (suitable) models. Overall, the algorithm yielded adequate performing neural networks despite the absence of the commonly required human intervention for adjustments.</p>
<p>It has been previously reported that different neural network hyperparameters can be obtained from network search algorithms for a given task (<xref ref-type="bibr" rid="ref6">Bellot et al., 2018</xref>; <xref ref-type="bibr" rid="ref22">Huang et al., 2020</xref>). The neural networks selected by the AutoML algorithm presented idiosyncrasies within replications of the same scenario (File S1). The lack of similarity between structures might arise from the ability of AutoML to adapt the network to the dataset (<xref ref-type="bibr" rid="ref23">Jin et al., 2019</xref>), which changes due to sampling in repeated validation. <xref ref-type="bibr" rid="ref22">Huang et al. (2020)</xref> suggest this event to be a consequence of insufficient data, but further confirmation is required. Additionally, this may also be associated with the sampling nature of the hyperparameter search system (<xref ref-type="bibr" rid="ref23">Jin et al., 2019</xref>). Despite the inconsistency between structures, systematic regularities are suggested by the within scenario low standard deviation of the estimated metrics (<xref rid="tab1" ref-type="table">Tables 1</xref> and <xref rid="tab2" ref-type="table">2</xref>). Hence, the networks might be capturing similar features, yielding consistent predictions. This has relevant implications for the choice of (deep) neural networks, meaning that distinct but adequate network structures result in similar outcomes.</p>
<p>Further observations can be drawn from the chosen network structures: (i) Although limited, the cases where structures did match (within and between scenarios) suggest that: type of task (regression or classification) is determinant over structure since most matches were of the same type; matches across traits were common, suggesting that similar sources of information might have been captured, which is probably intrinsically associated to the genetic correlation between PH and GY in maize. (ii) Also, when images were used as the input for prediction, augmentation procedures (e.g., resize, flip, rotation) were allocated in the structure of about half of the chosen models despite the spatial structure in the genomic images created by the decomposition performed by <italic>DeepInsight</italic>; further inferences on this matter would require studying the implications of such procedures to the original images, which is not in the scope of this study. (iii) At last, the depth and number of parameters of the networks within scenarios were highly variable for both MLP and CNN, suggesting that simple architectures were as effective as the more complex ones. Simpler models also have the advantage of being generally quicker to train (<xref ref-type="bibr" rid="ref45">van Dijk et al., 2021</xref>). (iv) Regarding overfitting, which is the tendency of a model to perform well on training but not on unseen data (<xref ref-type="bibr" rid="ref45">van Dijk et al., 2021</xref>), some differences in performance could be observed between inner training, inner validation, and outer validation sets, but further investigation would be required to determine their extent and consequences. The dropout regularization, temporarily setting a percentage of random neurons to zero (<xref ref-type="bibr" rid="ref41">Srivastava et al., 2014</xref>), was present on 2/3 of the chosen models, presumably acting on the overfitting issue (<xref ref-type="bibr" rid="ref32">Montesinos-L&#x00F3;pez et al., 2018b</xref>).</p>
</sec>
<sec id="sec26">
<title>Further Considerations</title>
<p>Overall, based on the empirical and experimental evidence, neural networks are especially competitive under the presence of strong nonlinear factors and interactions and hidden relationships between pieces of information. Accordingly, it is also dependent on the population type (e.g., lines or hybrids) and the consequent, non-mutually exclusive, genetic architecture of the trait (<xref ref-type="bibr" rid="ref6">Bellot et al., 2018</xref>; <xref ref-type="bibr" rid="ref1">Abdollahi-Arpanahi et al., 2020</xref>). The performance of NN is certainly conditioned on the choice of hyperparameters (<xref ref-type="bibr" rid="ref6">Bellot et al., 2018</xref>; <xref ref-type="bibr" rid="ref48">Zingaretti et al., 2020</xref>) and neural network type (MLP or CNN). It depends on how the input data is processed before prediction; consequently, special attention should be given to this step since valuable information could be lost. Also, it is presumably dependent on the number of samples and the sample to parameter ratio (<xref ref-type="bibr" rid="ref31">Montesinos-L&#x00F3;pez et al., 2018a</xref>; <xref ref-type="bibr" rid="ref4">Azodi et al., 2019</xref>; <xref ref-type="bibr" rid="ref37">P&#x00E9;rez-Enciso and Zingaretti, 2019</xref>; <xref ref-type="bibr" rid="ref1">Abdollahi-Arpanahi et al., 2020</xref>). Therefore, it is the scientist&#x2019;s discretion to test and identify the best performing method for their task. To this day, the only identified consistency regarding GP benchmarking is that no model performs best for all situations.</p>
<p>From experience, inferences on using images for GP could be drawn. In the original work by <xref ref-type="bibr" rid="ref39">Sharma et al. (2019)</xref>, <italic>DeepInsight</italic> was used for transforming RNA-seq, text, and artificial datasets into images. Our work is the first to apply such methodology in a GP context, and it is noteworthy that: (1) the algorithm can create images of different sizes. Image size, which is a hyperparameter, should be adapted to the available dataset and computational power. With the increasing size of the genomic matrix, there is a greater chance that a considerable amount of information would be lost as correlated markers would be tightly grouped, so larger images should be used (<xref ref-type="bibr" rid="ref39">Sharma et al., 2019</xref>). Additionally, increasing the size of images consequently increases the number of parameters estimated in the neural network, requiring greater computational power. In this work, using 120 by 120 images seemed to be an adequate fit for ~30,000 genomic markers; (2) different dimensionality reduction techniques can be used: t-SNE and kPCA are implemented in the algorithm, but any other of interest can be implemented; further testing should elaborate on this matter; (3) images can have multiple layers: neural networks can model linear and nonlinear relationships between neurons, including other effects and layers of data, such as dominance, epistasis, g&#x2009;&#x00D7;&#x2009;e, transcriptome, and so on.; (4) the cost&#x2013;benefit in terms of predictive gain and additional work, the use of images as input is arguable. Nevertheless, the methodology&#x2019;s potential for GP is unprecedented; (5) simulations should provide new valuable and unbiased information.</p>
<p>At last, we discussed two prediction alternatives: regression and classification. Under the regression context, MLP and CNN presented competitive results. Under the classification context, we expected better performances. Nevertheless, we believe that the latter has great potential for plant breeding since it simplifies the pipeline. Neural networks are self-adaptable and aimed at prediction alone. This statement implies that understanding and exposing the events underlying the relationship between phenotypes and genotypes are not of particular interest but could be done if necessary (<xref ref-type="bibr" rid="ref5">Azodi et al., 2020</xref>). This also implies that limited genetic knowledge of the trait is not a constraint for prediction. Coupled with a simpler processing, direct classification opens new possibilities regarding selecting traits where the ideotype points to intermediate phenotypes, e.g., plant height, ear height, and flowering time (under some circumstances) in maize. Hence, we believe this methodology deserves attention since it could further enhance the GP pipeline in breeding programs.</p>
</sec>
</sec>
<sec id="sec27" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref rid="sec30" ref-type="sec">Supplementary Material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="sec28">
<title>Author Contributions</title>
<p>GG elaborated on the hypothesis, conducted the analyses, and wrote the manuscript. RF-N, HC, RY, FS, JC, OM-L, and CG contributed to interpreting the results and writing. All authors have read and approved the final manuscript.</p>
</sec>
<sec id="sec41" sec-type="funding-information">
<title>Funding</title>
<p>This work was financially supported by Coordena&#x00E7;&#x00E3;o de Aperfei&#x00E7;oamento de Pessoal de N&#x00ED;vel Superior &#x2013; Brasil (CAPES) - Finance Code 001 and Conselho Nacional de Desenvolvimento Cient&#x00ED;fico e Tecnol&#x00F3;gico (CNPq). Funda&#x00E7;&#x00E3;o de Amparo &#x00E0; Pesquisa do Estado de S&#x00E3;o Paulo (FAPESP) and the Bill and Melinda Gates Foundation (BMGF): Grant Number INV-003439 BMGF/FCDO for the financial support. Accelerating Genetic Gains in Maize and Wheat for Improved Livelihoods (AG2MW).</p>
</sec>
<sec id="conf1" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec31" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack>
<p>The Allogamous Plant Breeding Laboratory team (Luiz de Queiroz College of Agriculture, University of S&#x00E3;o Paulo, Brazil) thanks Coordena&#x00E7;&#x00E3;o de Aperfei&#x00E7;oamento de Pessoal de N&#x00ED;vel Superior - Brasil (CAPES) - Finance Code 001, Conselho Nacional de Desenvolvimento Cient&#x00ED;fico e Tecnol&#x00F3;gico (CNPq). The Bill and Melinda Gates Foundation (BMGF) for the financial support.</p>
</ack>
<sec id="sec30" sec-type="supplementary-material">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2022.845524/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fpls.2022.845524/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.docx" id="SM1" mimetype="application/vnd.openxmlformats" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Data_Sheet_1.docx" id="SM2" mimetype="application/vnd.openxmlformats" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Image_1.pdf" id="SM3" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abdollahi-Arpanahi</surname> <given-names>R.</given-names></name> <name><surname>Gianola</surname> <given-names>D.</given-names></name> <name><surname>Pe&#x00F1;agaricano</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>Deep learning versus parametric and ensemble methods for genomic prediction of complex phenotypes</article-title>. <source>Genet. Sel.</source> <volume>52</volume>, <fpage>12</fpage>&#x2013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12711-020-00531-z</pub-id>, PMID: <pub-id pub-id-type="pmid">32093611</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alves</surname> <given-names>F. C.</given-names></name> <name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Matias</surname> <given-names>F. I.</given-names></name> <name><surname>Vidotti</surname> <given-names>M. S.</given-names></name> <name><surname>Morosini</surname> <given-names>J. S.</given-names></name> <name><surname>Fritsche-Neto</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <article-title>Impact of the complexity of genotype by environment and dominance modeling on the predictive accuracy of maize hybrids in multi-environment prediction models</article-title>. <source>Euphytica</source> <volume>217</volume>:<fpage>37</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s10681-021-02779-y</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alves</surname> <given-names>F. C.</given-names></name> <name><surname>Granato</surname> <given-names>&#x00CD;. S. C.</given-names></name> <name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Lyra</surname> <given-names>D. H.</given-names></name> <name><surname>Fritsche-Neto</surname> <given-names>R.</given-names></name> <name><surname>de los Campos</surname> <given-names>G.</given-names></name></person-group> (<year>2019</year>). <article-title>Bayesian analysis and prediction of hybrid performance</article-title>. <source>Plant Methods</source> <volume>15</volume>:<fpage>14</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13007-019-0388-x</pub-id>, PMID: <pub-id pub-id-type="pmid">30774704</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Azodi</surname> <given-names>C. B.</given-names></name> <name><surname>Bolger</surname> <given-names>E.</given-names></name> <name><surname>Mccarren</surname> <given-names>A.</given-names></name> <name><surname>Roantree</surname> <given-names>M.</given-names></name> <name><surname>Shiu</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>Benchmarking parametric and machine learning models for genomic prediction of complex traits</article-title>. <source>G3</source> <volume>9</volume>, <fpage>3691</fpage>&#x2013;<lpage>3702</lpage>. doi: <pub-id pub-id-type="doi">10.1534/g3.119.400498</pub-id>, PMID: <pub-id pub-id-type="pmid">31533955</pub-id></citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Azodi</surname> <given-names>C. B.</given-names></name> <name><surname>Tang</surname> <given-names>J.</given-names></name> <name><surname>Shiu</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>Opening the black box : interpretable machine learning for geneticists</article-title>. <source>Trends Genet.</source> <volume>36</volume>, <fpage>442</fpage>&#x2013;<lpage>455</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tig.2020.03.005</pub-id>, PMID: <pub-id pub-id-type="pmid">32396837</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bellot</surname> <given-names>P.</given-names></name> <name><surname>de los Campos</surname> <given-names>G.</given-names></name> <name><surname>P&#x00E9;rez-Enciso</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>Can deep learning improve genomic prediction of complex human traits?</article-title> <source>Genetics</source> <volume>210</volume>, <fpage>809</fpage>&#x2013;<lpage>819</lpage>. doi: <pub-id pub-id-type="doi">10.1534/genetics.118.301298</pub-id>, PMID: <pub-id pub-id-type="pmid">30171033</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bernardo</surname> <given-names>R.</given-names></name></person-group> (<year>1994</year>). <article-title>Prediction of maize single-cross performance using RFLPs and information from related hybrids</article-title>. <source>Crop Sci.</source> <volume>34</volume>:<fpage>20</fpage>, &#x2013;<lpage>25</lpage>. doi: <pub-id pub-id-type="doi">10.2135/cropsci1994.0011183X003400010003x</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blondel</surname> <given-names>M.</given-names></name> <name><surname>Onogi</surname> <given-names>A.</given-names></name> <name><surname>Iwata</surname> <given-names>H.</given-names></name> <name><surname>Ueda</surname> <given-names>N.</given-names></name></person-group> (<year>2015</year>). <article-title>A ranking approach to genomic selection</article-title>. <source>PLoS One</source> <volume>10</volume>:<fpage>e0128570</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0128570</pub-id>, PMID: <pub-id pub-id-type="pmid">26068103</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>C. C.</given-names></name> <name><surname>Chow</surname> <given-names>C. C.</given-names></name> <name><surname>Tellier</surname> <given-names>L. C.</given-names></name> <name><surname>Vattikuti</surname> <given-names>S.</given-names></name> <name><surname>Purcell</surname> <given-names>S. M.</given-names></name> <name><surname>Lee</surname> <given-names>J. J.</given-names></name></person-group> (<year>2015</year>). <article-title>Second-generation PLINK: rising to the challenge of larger and richer datasets</article-title>. <source>Gigascience</source> <volume>4</volume>:<fpage>7</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s13742-015-0047-8</pub-id>, PMID: <pub-id pub-id-type="pmid">25722852</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Costa-Neto</surname> <given-names>G.</given-names></name> <name><surname>Fritsche-Neto</surname> <given-names>R.</given-names></name> <name><surname>Crossa</surname> <given-names>J.</given-names></name></person-group> (<year>2021</year>). <article-title>Nonlinear kernels, dominance, and envirotyping data increase the accuracy of genome-based prediction in multi-environment trials</article-title>. <source>Heredity</source> <volume>126</volume>, <fpage>92</fpage>&#x2013;<lpage>106</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41437-020-00353-1</pub-id>, PMID: <pub-id pub-id-type="pmid">32855544</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>de Los Campos</surname> <given-names>G.</given-names></name> <name><surname>Gianola</surname> <given-names>D.</given-names></name> <name><surname>Rosa</surname> <given-names>G. J.</given-names></name></person-group> (<year>2009</year>). <article-title>Reproducing kernel Hilbert spaces regression: a general framework for genetic evaluation</article-title>. <source>J. Anim. Sci.</source> <volume>87</volume>, <fpage>1883</fpage>&#x2013;<lpage>1887</lpage>. doi: <pub-id pub-id-type="doi">10.2527/jas.2008-1259</pub-id>, PMID: <pub-id pub-id-type="pmid">19213705</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Fern&#x00E1;ndez</surname> <given-names>A.</given-names></name> <name><surname>Garc&#x00ED;a</surname> <given-names>S.</given-names></name> <name><surname>Herrera</surname> <given-names>F.</given-names></name></person-group> (<year>2011</year>). &#x201C;<article-title>Addressing the classification with imbalanced data: open problems and new challenges on class distribution,</article-title>&#x201D; in <source>Hybrid Artificial Intelligent Systems.</source> eds. <person-group person-group-type="editor"><name><surname>Corchado</surname> <given-names>E.</given-names></name> <name><surname>Kurzy&#x0144;ski</surname> <given-names>M.</given-names></name> <name><surname>Wo&#x017A;niak</surname> <given-names>M.</given-names></name></person-group> (<publisher-loc>Berlin Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>10</lpage>.</citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Feurer</surname> <given-names>M.</given-names></name> <name><surname>Klein</surname> <given-names>A.</given-names></name> <name><surname>Eggensperger</surname> <given-names>K.</given-names></name> <name><surname>Springenberg</surname> <given-names>J. T.</given-names></name> <name><surname>Blum</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Efficient and robust automated machine learning</article-title>. <source>Adv. Neural Info. Process. Syst.</source></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fritsche-Neto</surname> <given-names>R.</given-names></name> <name><surname>Akdemir</surname> <given-names>D.</given-names></name> <name><surname>Jannink</surname> <given-names>J.-L.</given-names></name></person-group> (<year>2018</year>). <article-title>Accuracy of genomic selection to predict maize single-crosses obtained through different mating designs</article-title>. <source>Theor. Appl. Genet.</source> <volume>131</volume>, <fpage>1153</fpage>&#x2013;<lpage>1162</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00122-018-3068-8</pub-id>, PMID: <pub-id pub-id-type="pmid">29445844</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fritsche-Neto</surname> <given-names>R.</given-names></name> <name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>de Mendon&#x00E7;a</surname> <given-names>L. F.</given-names></name> <name><surname>Vidotti</surname> <given-names>M. S.</given-names></name> <name><surname>Matias</surname> <given-names>F. I.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>USP tropical maize hybrid panel</article-title>. <source>Mendeley Data</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.17632/tpcw383fkm.3</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Alves</surname> <given-names>F. C.</given-names></name> <name><surname>Morosini</surname> <given-names>J. S.</given-names></name> <name><surname>Fritsche-Neto</surname> <given-names>R.</given-names></name></person-group> (<year>2020a</year>). <article-title>On the usefulness of parental lines GWAS for predicting low heritability traits in tropical maize hybrids</article-title>. <source>PLoS One</source> <volume>15</volume>, <fpage>e0228724</fpage>&#x2013;<lpage>e0228715</lpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0228724</pub-id>, PMID: <pub-id pub-id-type="pmid">32032385</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Horne</surname> <given-names>D. W.</given-names></name> <name><surname>Fritsche-neto</surname> <given-names>R.</given-names></name> <name><surname>Rooney</surname> <given-names>W. L.</given-names></name></person-group> (<year>2020b</year>). <article-title>Optimization of UAS-based high-throughput phenotyping to estimate plant health and grain yield in sorghum</article-title>. <source>Plant Phenom. J.</source> <volume>3</volume>, <fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.1002/ppj2.20010</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Lyra</surname> <given-names>D. H.</given-names></name> <name><surname>Alves</surname> <given-names>F. C.</given-names></name> <name><surname>Granato</surname> <given-names>&#x00CD;. S. C.</given-names></name> <name><surname>e Sousa</surname> <given-names>M. B.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Impact of phenotypic correction method and missing phenotypic data on genomic prediction of maize hybrids</article-title>. <source>Crop Sci.</source> <volume>58</volume>, <fpage>1481</fpage>&#x2013;<lpage>1491</lpage>. doi: <pub-id pub-id-type="doi">10.2135/cropsci2017.07.0459</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Gilmour</surname> <given-names>A. R.</given-names></name> <name><surname>Gogel</surname> <given-names>B. J.</given-names></name> <name><surname>Cullis</surname> <given-names>B. R.</given-names></name> <name><surname>Thompson</surname> <given-names>R.</given-names></name></person-group> (<year>2009</year>). <source>ASReml User Guide Release 3.0</source>, <publisher-loc>Hemel Hempstead</publisher-loc>: <publisher-name>VSN International</publisher-name>.</citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gonz&#x00E1;lez-Camacho</surname> <given-names>J. M.</given-names></name> <name><surname>Crossa</surname> <given-names>J.</given-names></name> <name><surname>P&#x00E9;rez-Rodr&#x00ED;guez</surname> <given-names>P.</given-names></name> <name><surname>Ornella</surname> <given-names>L.</given-names></name> <name><surname>Gianola</surname> <given-names>D.</given-names></name></person-group> (<year>2016</year>). <article-title>Genome-enabled prediction using probabilistic neural network classifiers</article-title>. <source>BMC Genomics</source> <volume>17</volume>, <fpage>208</fpage>&#x2013;<lpage>216</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12864-016-2553-1</pub-id>, PMID: <pub-id pub-id-type="pmid">26956885</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Granato</surname> <given-names>I. S. C.</given-names></name> <name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>de Oliveira Couto</surname> <given-names>E. G.</given-names></name> <name><surname>e Souza</surname> <given-names>M. B.</given-names></name> <name><surname>Mendon&#x00E7;a</surname> <given-names>L. F.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>snpReady: a tool to assist breeders in genomic analysis</article-title>. <source>Mol. Breed.</source> <volume>38</volume>:<fpage>102</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s11032-018-0844-8</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>G. H.</given-names></name> <name><surname>Lin</surname> <given-names>C. H.</given-names></name> <name><surname>Cai</surname> <given-names>Y. R.</given-names></name> <name><surname>Chen</surname> <given-names>T. B.</given-names></name> <name><surname>Hsu</surname> <given-names>S. Y.</given-names></name> <name><surname>Lu</surname> <given-names>N. H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Multiclass machine learning classification of functional brain images for Parkinson&#x2019;s disease stage prediction</article-title>. <source>Stat. Anal. Data Min.</source> <volume>13</volume>, <fpage>508</fpage>&#x2013;<lpage>523</lpage>. doi: <pub-id pub-id-type="doi">10.1002/sam.11480</pub-id></citation></ref>
<ref id="ref23"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Jin</surname> <given-names>H.</given-names></name> <name><surname>Song</surname> <given-names>Q.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name></person-group> (<year>2019</year>). Auto-Keras : an efficient neural architecture search system: 1946&#x2013;1956, July 2019.</citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kotthoff</surname> <given-names>L.</given-names></name> <name><surname>Thornton</surname> <given-names>C.</given-names></name> <name><surname>Hoos</surname> <given-names>H. H.</given-names></name> <name><surname>Hutter</surname> <given-names>F.</given-names></name> <name><surname>Leyton-Brown</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>Auto-WEKA 2.0: automatic model selection and hyperparameter optimization in WEKA</article-title>. <source>J. Mach. Learn. Res.</source> doi: <pub-id pub-id-type="doi">10.1007/978-3-030-05318-5_4</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lyra</surname> <given-names>D. H.</given-names></name> <name><surname>de Freitas Mendon&#x00E7;a</surname> <given-names>L.</given-names></name> <name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Alves</surname> <given-names>F. C.</given-names></name> <name><surname>Granato</surname> <given-names>&#x00CD;. S. C.</given-names></name> <name><surname>Fritsche-Neto</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Multi-trait genomic prediction for nitrogen response indices in tropical maize hybrids</article-title>. <source>Mol. Breed.</source> <volume>37</volume>:<fpage>80</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s11032-017-0681-1</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lyra</surname> <given-names>D. H.</given-names></name> <name><surname>Granato</surname> <given-names>&#x00CD;. S. C.</given-names></name> <name><surname>Morais</surname> <given-names>P. P. P.</given-names></name> <name><surname>Alves</surname> <given-names>F. C.</given-names></name> <name><surname>dos Santos</surname> <given-names>A. R. M.</given-names></name> <name><surname>Yu</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Controlling population structure in the genomic prediction of tropical maize hybrids</article-title>. <source>Mol. Breed.</source> <volume>38</volume>:<fpage>126</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s11032-018-0882-2</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>W.</given-names></name> <name><surname>Qiu</surname> <given-names>Z.</given-names></name> <name><surname>Song</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Cheng</surname> <given-names>Q.</given-names></name> <name><surname>Zhai</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>A deep convolutional neural network approach for predicting phenotypes from genotypes</article-title>. <source>Planta</source> <volume>248</volume>, <fpage>1307</fpage>&#x2013;<lpage>1318</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00425-018-2976-9</pub-id>, PMID: <pub-id pub-id-type="pmid">30101399</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Matias</surname> <given-names>F. I.</given-names></name> <name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Correia Granato</surname> <given-names>I. S.</given-names></name> <name><surname>Fritsche-Neto</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Genomic prediction of Autogamous and Allogamous plants by SNPs and haplotypes</article-title>. <source>Crop Sci.</source> <volume>57</volume>, <fpage>2951</fpage>&#x2013;<lpage>2958</lpage>. doi: <pub-id pub-id-type="doi">10.2135/cropsci2017.01.0022</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Meuwissen</surname> <given-names>T. H. E.</given-names></name> <name><surname>Hayes</surname> <given-names>B. J.</given-names></name> <name><surname>Goddard</surname> <given-names>M. E.</given-names></name></person-group> (<year>2001</year>). <article-title>Prediction of Total genetic value using genome-wide dense marker maps</article-title>. <source>Genetics</source> <volume>157</volume>, <fpage>1819</fpage>&#x2013;<lpage>1829</lpage>. doi: <pub-id pub-id-type="doi">10.1093/genetics/157.4.1819</pub-id>, PMID: <pub-id pub-id-type="pmid">11290733</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>O. A.</given-names></name> <name><surname>Mart&#x00ED;n-Vallejo</surname> <given-names>J.</given-names></name> <name><surname>Crossa</surname> <given-names>J.</given-names></name> <name><surname>Gianola</surname> <given-names>D.</given-names></name> <name><surname>Hern&#x00E1;ndez-Su&#x00E1;rez</surname> <given-names>C. M.</given-names></name> <name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>A benchmarking Between deep learning, support vector machine and Bayesian threshold best linear unbiased prediction for predicting ordinal traits in plant breeding</article-title>. <source>G3</source> <volume>9</volume>, <fpage>601</fpage>&#x2013;<lpage>618</lpage>. doi: <pub-id pub-id-type="doi">10.1534/g3.118.200998</pub-id>, PMID: <pub-id pub-id-type="pmid">30593512</pub-id></citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>O. A.</given-names></name> <name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>A.</given-names></name> <name><surname>Crossa</surname> <given-names>J.</given-names></name> <name><surname>Gianola</surname> <given-names>D.</given-names></name> <name><surname>Hern&#x00E1;ndez-Su&#x00E1;rez</surname> <given-names>C. M.</given-names></name> <name><surname>Mart&#x00ED;n-Vallejo</surname> <given-names>J.</given-names></name></person-group> (<year>2018a</year>). <article-title>Multi-trait, multi-environment deep learning modeling for genomic-enabled prediction of plant traits</article-title>. <source>G3</source> <volume>8</volume>, <fpage>3829</fpage>&#x2013;<lpage>3840</lpage>. doi: <pub-id pub-id-type="doi">10.1534/g3.118.200728</pub-id>, PMID: <pub-id pub-id-type="pmid">30291108</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>A.</given-names></name> <name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>O. A.</given-names></name> <name><surname>Gianola</surname> <given-names>D.</given-names></name> <name><surname>Crossa</surname> <given-names>J.</given-names></name> <name><surname>Hern&#x00E1;ndez-Su&#x00E1;rez</surname> <given-names>C. M.</given-names></name></person-group> (<year>2018b</year>). <article-title>Multi-environment genomic prediction of plant traits using deep learners with dense architecture</article-title>. <source>G3</source> <volume>8</volume>, <fpage>3813</fpage>&#x2013;<lpage>3828</lpage>. doi: <pub-id pub-id-type="doi">10.1534/g3.118.200740</pub-id>, PMID: <pub-id pub-id-type="pmid">30291107</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>O. A.</given-names></name> <name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>A.</given-names></name> <name><surname>Mosqueda-Gonzalez</surname> <given-names>B. A.</given-names></name> <name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>J. C.</given-names></name> <name><surname>Crossa</surname> <given-names>J.</given-names></name> <name><surname>Ramirez</surname> <given-names>N. L.</given-names></name> <etal/></person-group>. (<year>2021a</year>). <article-title>A zero altered Poisson random forest model for genomic-enabled prediction (E. Akhunov, editor)</article-title>. <source>G3</source> <volume>11</volume>. doi: <pub-id pub-id-type="doi">10.1093/g3journal/jkaa057</pub-id>, PMID: <pub-id pub-id-type="pmid">33693599</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>O. A.</given-names></name> <name><surname>Montesinos-L&#x00F3;pez</surname> <given-names>A.</given-names></name> <name><surname>P&#x00E9;rez-Rodr&#x00ED;guez</surname> <given-names>P.</given-names></name> <name><surname>Barr&#x00F3;n-L&#x00F3;pez</surname> <given-names>J. A.</given-names></name> <name><surname>Martini</surname> <given-names>J. W. R.</given-names></name> <name><surname>Fajardo-Flores</surname> <given-names>S. B.</given-names></name> <etal/></person-group>. (<year>2021b</year>). <article-title>A review of deep learning applications for genomic selection</article-title>. <source>BMC Genomics</source> <volume>22</volume>, <fpage>19</fpage>&#x2013;<lpage>23</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s12864-020-07319-x</pub-id>, PMID: <pub-id pub-id-type="pmid">33407114</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Morosini</surname> <given-names>J. S.</given-names></name> <name><surname>de Mendon&#x00E7;a</surname> <given-names>L. F.</given-names></name> <name><surname>Lyra</surname> <given-names>D. H.</given-names></name> <name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Vidotti</surname> <given-names>M. S.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Association mapping for traits related to nitrogen use efficiency in tropical maize lines under field conditions</article-title>. <source>Plant Soil</source> <volume>421</volume>, <fpage>453</fpage>&#x2013;<lpage>463</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11104-017-3479-3</pub-id></citation></ref>
<ref id="ref36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ornella</surname> <given-names>L.</given-names></name> <name><surname>P&#x00E9;rez</surname> <given-names>P.</given-names></name> <name><surname>Tapia</surname> <given-names>E.</given-names></name> <name><surname>Gonz&#x00E1;lez-Camacho</surname> <given-names>J. M.</given-names></name> <name><surname>Burgue&#x00F1;o</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Genomic-enabled prediction with classification algorithms</article-title>. <source>Heredity</source> <volume>112</volume>, <fpage>616</fpage>&#x2013;<lpage>626</lpage>. doi: <pub-id pub-id-type="doi">10.1038/hdy.2013.144</pub-id>, PMID: <pub-id pub-id-type="pmid">24424163</pub-id></citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>P&#x00E9;rez-Enciso</surname> <given-names>M.</given-names></name> <name><surname>Zingaretti</surname> <given-names>L. M.</given-names></name></person-group> (<year>2019</year>). <article-title>A guide for using deep learning for complex trait genomic prediction</article-title>. <source>Genes</source> <volume>10</volume>, <fpage>1</fpage>&#x2013;<lpage>19</lpage>. doi: <pub-id pub-id-type="doi">10.3390/genes10070553</pub-id>, PMID: <pub-id pub-id-type="pmid">31330861</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pook</surname> <given-names>T.</given-names></name> <name><surname>Freudenthal</surname> <given-names>J.</given-names></name> <name><surname>Korte</surname> <given-names>A.</given-names></name> <name><surname>Simianer</surname> <given-names>H.</given-names></name></person-group> (<year>2020</year>). <article-title>Using local convolutional neural networks for genomic prediction</article-title>. <source>Front. Genet.</source> <volume>11</volume>. doi: <pub-id pub-id-type="doi">10.3389/fgene.2020.561497</pub-id>, PMID: <pub-id pub-id-type="pmid">33281867</pub-id></citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sharma</surname> <given-names>A.</given-names></name> <name><surname>Vans</surname> <given-names>E.</given-names></name> <name><surname>Shigemizu</surname> <given-names>D.</given-names></name> <name><surname>Boroevich</surname> <given-names>K. A.</given-names></name></person-group> (<year>2019</year>). <article-title>DeepInsight: a methodology to transform a non-image data to an image for convolution neural network architecture</article-title>. <source>Sci. Rep.</source> <volume>9</volume>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-019-47765-6</pub-id>, PMID: <pub-id pub-id-type="pmid">31388036</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sousa</surname> <given-names>M. B.</given-names></name> <name><surname>Galli</surname> <given-names>G.</given-names></name> <name><surname>Lyra</surname> <given-names>D. H.</given-names></name> <name><surname>Granato</surname> <given-names>&#x00CD;. S. C.</given-names></name> <name><surname>Matias</surname> <given-names>F. I.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Increasing accuracy and reducing costs of genomic prediction by marker selection</article-title>. <source>Euphytica</source> <volume>215</volume>:<fpage>18</fpage>. doi: <pub-id pub-id-type="doi">10.1007/s10681-019-2339-z</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Srivastava</surname> <given-names>N.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name> <name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name></person-group> (<year>2014</year>). <article-title>Dropout: A simple way to prevent neural networks from overfitting</article-title>. <source>J. Mach. Learn. Res.</source></citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Trevisan</surname> <given-names>R. G.</given-names></name> <name><surname>P&#x00E9;rez</surname> <given-names>O.</given-names></name> <name><surname>Schmitz</surname> <given-names>N.</given-names></name> <name><surname>Diers</surname> <given-names>B. W.</given-names></name> <name><surname>Nicolas</surname> <given-names>F.</given-names></name></person-group> (<year>2020</year>). <article-title>High-throughput Phenotyping of soybean maturity using time series UAV imagery and convolutional</article-title>. <source>Neural Netw.</source> doi: <pub-id pub-id-type="doi">10.20944/preprints202009.0458.v1</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Truong</surname> <given-names>A.</given-names></name> <name><surname>Walters</surname> <given-names>A.</given-names></name> <name><surname>Goodsitt</surname> <given-names>J.</given-names></name> <name><surname>Hines</surname> <given-names>K.</given-names></name> <name><surname>Bruss</surname> <given-names>C. B.</given-names></name> <etal/></person-group>. (<year>2019</year>). Towards automated machine learning: evaluation and comparison of AutoML approaches and tools. Proceedings &#x2013; International Conference on Tools with Artificial Intelligence, ICTAI, November 4, 2019.</citation></ref>
<ref id="ref44"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Unterseer</surname> <given-names>S.</given-names></name> <name><surname>Bauer</surname> <given-names>E.</given-names></name> <name><surname>Haberer</surname> <given-names>G.</given-names></name> <name><surname>Seidel</surname> <given-names>M.</given-names></name> <name><surname>Knaak</surname> <given-names>C.</given-names></name> <name><surname>Ouzunova</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>A powerful tool for genome analysis in maize: development and evaluation of the high density 600 k SNP genotyping array</article-title>. <source>BMC Genomics</source> <volume>15</volume>:<fpage>823</fpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2164-15-823</pub-id>, PMID: <pub-id pub-id-type="pmid">25266061</pub-id></citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>van Dijk</surname> <given-names>A. D. J.</given-names></name> <name><surname>Kootstra</surname> <given-names>G.</given-names></name> <name><surname>Kruijer</surname> <given-names>W.</given-names></name> <name><surname>de Ridder</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Machine learning in plant science and plant breeding</article-title>. <source>iScience</source> <volume>24</volume>:<fpage>101890</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.isci.2020.101890</pub-id>, PMID: <pub-id pub-id-type="pmid">33364579</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>VanRaden</surname> <given-names>P. M.</given-names></name></person-group> (<year>2008</year>). <article-title>Efficient methods to compute genomic predictions</article-title>. <source>J. Dairy Sci.</source> <volume>91</volume>, <fpage>4414</fpage>&#x2013;<lpage>4423</lpage>. doi: <pub-id pub-id-type="doi">10.3168/jds.2007-0980</pub-id>, PMID: <pub-id pub-id-type="pmid">18946147</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wimmer</surname> <given-names>V.</given-names></name> <name><surname>Albrecht</surname> <given-names>T.</given-names></name> <name><surname>Auinger</surname> <given-names>H.-J.</given-names></name> <name><surname>Schon</surname> <given-names>C.-C.</given-names></name></person-group> (<year>2012</year>). <article-title>Synbreed: a framework for the analysis of genomic prediction data using R</article-title>. <source>Bioinformatics</source> <volume>28</volume>, <fpage>2086</fpage>&#x2013;<lpage>2087</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bts335</pub-id>, PMID: <pub-id pub-id-type="pmid">22689388</pub-id></citation></ref>
<ref id="ref48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zingaretti</surname> <given-names>L. M.</given-names></name> <name><surname>Gezan</surname> <given-names>S. A.</given-names></name> <name><surname>Ferr&#x00E3;o</surname> <given-names>L. F. V.</given-names></name> <name><surname>Osorio</surname> <given-names>L. F.</given-names></name> <name><surname>Monfort</surname> <given-names>A.</given-names></name> <name><surname>Mu&#x00F1;oz</surname> <given-names>P. R.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Exploring deep learning for complex trait genomic prediction in Polyploid outcrossing species</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>, <fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2020.00025</pub-id>, PMID: <pub-id pub-id-type="pmid">32117371</pub-id></citation></ref></ref-list>
</back>
</article>