<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1537990</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Analyses of crop yield dynamics and the development of a multimodal neural network prediction model with G&#xd7;E&#xd7;M interactions</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Sajid</surname>
<given-names>Saiara Samira</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1591117/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Khalilzadeh</surname>
<given-names>Zahra</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2516765/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Lizhi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/462330/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hu</surname>
<given-names>Guiping</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Iowa State University, Industrial Manufacturing &amp; Systems Engineering</institution>, <addr-line>Ames, IA</addr-line>,&#xa0;<country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Oklahoma State University, Industrial Engineering &amp; Management</institution>, <addr-line>Stillwater, OK</addr-line>,&#xa0;<country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Huajian Liu, University of Adelaide, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Muhammad Munir, King Faisal University, Saudi Arabia</p>
<p>Benjamin Kwapong Osibo, Nanjing University of Information Science and Technology, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Saiara Samira Sajid, <email xlink:href="mailto:sajids@iastate.edu">sajids@iastate.edu</email>; <email xlink:href="mailto:saiarasamiras@gmail.com">saiarasamiras@gmail.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>31</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1537990</elocation-id>
<history>
<date date-type="received">
<day>02</day>
<month>12</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Sajid, Khalilzadeh, Wang and Hu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Sajid, Khalilzadeh, Wang and Hu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>This study investigated how genotype, environment, and management (G&#xd7;E&#xd7;M) interactions influence yield and highlight the importance of accurate, early yield predictions for effective farm management and enhancing food security. We developed a yield prediction model capable of determining field-level outputs based on comprehensive data inputs, including genotype, spatial, temporal, environmental, and management factors. Among tested models&#x2014;LASSO, Random Forest, XGBoost, single-modal CNN-DNN, and multimodal CNN-DNN&#x2014;the multimodal CNN-DNN ensembled with XGBoost demonstrated superior performance. Applied to the G2F dataset covering 21 states from 2014 to 2021 across various treatments (i.e., standard, drought, irrigation, disease trials), the model excelled particularly in stable historical yield settings (RMSE 2.36 Mg/ha for standard treatment) with an overall RMSE of 2.45 Mg/ha. Additionally, we introduced an empirical tool for identifying high-yield hybrids suitable for standard and challenging conditions. Exploratory analysis confirmed that crop yields vary greatly by hybrid and location interaction and that late planting generally yields less than standard timing. Customized management strategies based on specific local and hybrid conditions are crucial for optimal yield outcomes.</p>
</abstract>
<kwd-group>
<kwd>genotype</kwd>
<kwd>planting date</kwd>
<kwd>high-yield hybrid classification</kwd>
<kwd>precision farming</kwd>
<kwd>multimodal CNN-DNN</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Science Foundation<named-content content-type="fundref-id">10.13039/100000001</named-content>
</contract-sponsor>
<counts>
<fig-count count="15"/>
<table-count count="3"/>
<equation-count count="7"/>
<ref-count count="71"/>
<page-count count="20"/>
<word-count count="8374"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Technical Advances in Plant Science</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>To ensure global food security, establishing a sustainable food supply chain is essential. A major factor in achieving this sustainability is enhancing crop productivity and developing precise crop prediction models. Crop yields are shaped by multiple influences, such as environmental conditions, crop hybrids, and farming practices. Gaining a deeper understanding of these factors is key to improving decision-making and improving productivity. Moreover, precise yield predictions can aid informed management decisions throughout the growing season, guiding resource allocation and ultimately helping secure a reliable future food supply.</p>
<p>The timing of planting is an important farm management decision that greatly influences crop yield (<xref ref-type="bibr" rid="B62">Swanson and Wilhelm, 1996</xref>; <xref ref-type="bibr" rid="B35">Lauer et&#xa0;al., 1999</xref>; <xref ref-type="bibr" rid="B15">Darby and Lauer, 2002</xref>; <xref ref-type="bibr" rid="B1">Anapalli et&#xa0;al., 2005</xref>; <xref ref-type="bibr" rid="B70">Williams, 2006</xref>; <xref ref-type="bibr" rid="B67">Van Roekel and Coulter, 2011</xref>), while the optimal planting date varies based on crop hybrid (<xref ref-type="bibr" rid="B41">Masud Rana et&#xa0;al., 2024</xref>). The effect of the planting date also changes by location due to environmental factors, with some areas experiencing a more significant impact on yield due to planting decisions (<xref ref-type="bibr" rid="B49">Mrubata et&#xa0;al., 2024</xref>). Additionally, the planting date interacts with soil properties (<xref ref-type="bibr" rid="B5">Bollero et&#xa0;al., 1996</xref>) and fertilizer application (<xref ref-type="bibr" rid="B26">Hankinson et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B30">Kaiser et&#xa0;al., 2016</xref>), influencing yield response. Compared to early or planting at the right time, late planting results in higher yield reduction (<xref ref-type="bibr" rid="B48">Moseley et&#xa0;al., 2024</xref>). Hence, planting time recommendations should be customized to location and crop hybrid. In this study, we examined the influence of planting dates on yield across various geographic regions and weather conditions to develop a more comprehensive understanding of their impact on crop productivity.</p>
<p>Selecting crop hybrids that are resilient to diverse weather conditions and soil properties is essential, particularly for areas susceptible to extreme weather events like drought or disease outbreaks. Using available tools to understand genetic variation and support crop improvement can be complex and challenging (<xref ref-type="bibr" rid="B28">Heffner et&#xa0;al., 2009</xref>; <xref ref-type="bibr" rid="B57">Scheben et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B51">Nuccio et&#xa0;al., 2018</xref>). <xref ref-type="bibr" rid="B17">Dobermann et&#xa0;al. (2003)</xref> empirically classified yield and discovered that yield clusters account for 60-66% of the yield variability. Additionally, <xref ref-type="bibr" rid="B39">Maestrini and Basso (2018)</xref> demonstrated that historical yield distributions from locations with stable data can provide accurate yield predictions. Combining these concepts, our work developed an empirical hybrid classification method to identify hybrids that are well-suited for varying weather conditions and are straightforward to use for making management decisions.</p>
<p>The complex interaction among meteorology, soil characteristics, management decisions, and genomic traits of hybrids with crop yield makes the prediction task challenging. Recent advances in crop genomics have paved ways to access genotype data that provides insights into how these genetic factors shape crop characteristics (<xref ref-type="bibr" rid="B44">Mir et&#xa0;al., 2019</xref>). However, incorporating genotype data, among other factors, adds to the complexity of the task, primarily due to the high dimensionality of the data.</p>
<p>Machine learning (ML) models have demonstrated noteworthy success in handling high-dimensional data. These models are trained based on historical data to establish the mapping function that links the input variable to the output (<xref ref-type="bibr" rid="B42">Medsker and Jain, 2001</xref>). Their effectiveness in predicting crop yield has been substantiated by various studies (<xref ref-type="bibr" rid="B20">Everingham et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B12">Cunha et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B34">Kouadio et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B22">Filippi et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B58">Shahhosseini et&#xa0;al., 2020</xref>, <xref ref-type="bibr" rid="B59">2021a</xref>; <xref ref-type="bibr" rid="B2">Bali and Singla, 2021</xref>; <xref ref-type="bibr" rid="B54">Paudel et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B55">Sajid et&#xa0;al., 2022</xref>). These investigations highlight the ML model&#x2019;s ability to comprehend intricate relationships between yield, environment, and management.</p>
<p>Moreover, neural networks (NN) have also proven their efficacy in capturing the intricate relationship between weather, soil, management decisions, and yield (<xref ref-type="bibr" rid="B68">Wang et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B50">Nevavuori et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B19">Elavarasan and Durairaj Vincent, 2020</xref>; <xref ref-type="bibr" rid="B27">Haque et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B32">Khaki et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B60">Shahhosseini et&#xa0;al., 2021b</xref>; <xref ref-type="bibr" rid="B53">Oikonomidis et&#xa0;al., 2022</xref>). For example, <xref ref-type="bibr" rid="B60">Shahhosseini et&#xa0;al. (2021b)</xref> developed an ensemble convolutional neural network (CNN)-deep neural network (DNN) architecture to predict corn yield for the US corn belt using weather, soil, and management inputs. <xref ref-type="bibr" rid="B33">Kolipaka and Namburu (2024)</xref> proposed a heuristic approach integrating CNN-DNN and long-short-term memory (LSTM) for yield prediction across various datasets. However, these studies have primarily focused on crop yield&#x2019;s interaction with environmental and management aspects, often overlooking the incorporation of genotype data into their prediction models.</p>
<p>In addition to environmental (E) and management decisions (M), genotype (G) plays a pivotal role in crop yield (<xref ref-type="bibr" rid="B4">Beres et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B37">Lopez-Cruz et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B21">Fernandes et&#xa0;al., 2024</xref>). <xref ref-type="bibr" rid="B23">Gambin et&#xa0;al. (2016)</xref> discovered that G <inline-formula>
<mml:math display="inline" id="im1">
<mml:mo>&#xd7;</mml:mo>
</mml:math>
</inline-formula>E interactions significantly impact crop yield in a study conducted in Argentina. However, complexities arise, as genetic selection can result in traits unique to new varieties even under the same environmental conditions. The complexity further escalates when considering a wide range of environmental and management conditions (<xref ref-type="bibr" rid="B52">Oakey et&#xa0;al., 2016</xref>). Additionally, the high dimensionality of genotype data adds another layer of complexity to predicting G <inline-formula>
<mml:math display="inline" id="im2">
<mml:mo>&#xd7;</mml:mo>
</mml:math>
</inline-formula>E interactions.</p>
<p>Some studies have utilized ML and NN models alongside plant genetics to predict plant phenotype traits (<xref ref-type="bibr" rid="B38">Ma et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B47">Montesinos-L&#xf3;pez et&#xa0;al., 2018a</xref>; <xref ref-type="bibr" rid="B46">b</xref>; <xref ref-type="bibr" rid="B11">Crossa et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B25">Grinberg et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B61">Shook et&#xa0;al., 2021</xref>). Notably, <xref ref-type="bibr" rid="B31">Khaki and Wang (2019)</xref> employed a DNN architecture to predict yield from G <inline-formula>
<mml:math display="inline" id="im3">
<mml:mo>&#xd7;</mml:mo>
</mml:math>
</inline-formula>E interactions for new hybrids. A recent study by <xref ref-type="bibr" rid="B16">Dhaliwal and Williams (2024)</xref> demonstrated that Random Forest (RF) could predict yield for sweet corn at the field level using weather, spatial, temporal, and genetic data. ML models have demonstrated superior prediction accuracy compared to the genomic prediction tool, genomic best linear unbiased prediction (GBLUP). GBLUP lacks consideration for non-linear relations and exhibits limitations compared to ML models (<xref ref-type="bibr" rid="B14">Danilevicz et&#xa0;al., 2022</xref>). <xref ref-type="bibr" rid="B56">Sarzaeim and Mu&#xf1;oz-Arriola (2024)</xref> further combined genetic prediction models with global sensitivity analysis and highlighted weather as a key factor in yield forecasting. Moreover, <xref ref-type="bibr" rid="B69">Washburn et&#xa0;al. (2024)</xref> highlighted that diverse modeling approaches enhance predictability when considering G&#xd7;E&#xd7;M interactions.</p>
<p>In the context of using genotype data for phenotype predictions, NN have shown potential; however, their full capabilities remain largely unexplored (<xref ref-type="bibr" rid="B14">Danilevicz et&#xa0;al., 2022</xref>). Multimodal model architectures designed for multi-source data have shown improved predictability for high dimensional data (<xref ref-type="bibr" rid="B3">Baltrusaitis et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B40">Maimaitijiang et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B60">Shahhosseini et&#xa0;al., 2021b</xref>; <xref ref-type="bibr" rid="B43">Mia et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B71">Yu et&#xa0;al., 2024</xref>). In this study, we built a multimodal NN model to predict crop yield given weather, soil, environment, genotype, and location information.</p>
<p>This research utilizes the dataset provided in the G2F competition (<xref ref-type="bibr" rid="B36">Lima et&#xa0;al., 2023</xref>) and aims to examine the factors influencing agricultural productivity and predict yield by incorporating G&#xd7;E&#xd7;M interactions. The study includes a statistical analysis of factors impacting crop yield and the development of a prediction model. The analysis of hybrid maize yield across different environments in the first part is to understand qualitatively and quantitatively the G&#xd7;E&#xd7;M interactions, which motivated the design of our prediction model in the second part. The proposed multimodal NN architecture considers various elements such as weather, soil, environment, genotype, temporal and spatial factors to make field-level predictions. Additionally, this research seeks to understand the influence of these factors and to develop effective methodologies for enhancing decision-making to improve agricultural results. This research has three objectives: a) to explore historical data and identify factors influencing productivity; b) to create a hybrid selection tool for normal and extreme conditions (e.g., drought, disease outbreaks); and c) to develop a predictive model for different hybrids based on G <inline-formula>
<mml:math display="inline" id="im4">
<mml:mo>&#xd7;</mml:mo>
</mml:math>
</inline-formula>E <inline-formula>
<mml:math display="inline" id="im5">
<mml:mo>&#xd7;</mml:mo>
</mml:math>
</inline-formula>M interactions.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<p>This study aims to identify the prevailing factors impacting crop yield and assist decision-makers in agricultural production. To determine the factors influencing crop yield, an exploratory analysis was first conducted to gauge the effect of place, hybrid varieties, and approaches to management. Subsequently, to aid in hybrid selection, a tool was developed to identify hybrids that perform well even in extreme scenarios. Finally, a yield prediction model was constructed using genotype, meteorological data, soil data, and management practices to provide a reliable estimate of crop yield.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Data sources and preprocessing</title>
<p>The data used in this study were adapted from the G2F initiative (<xref ref-type="bibr" rid="B36">Lima et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B63">The Genomes To Fields Initiative, 2023</xref>), encompassing over 180,000 cornfield plots in 217 different environments. A detailed description of the data used in this study is provided in <xref ref-type="bibr" rid="B24">Genomes to Fields (2023)</xref>. The dataset comprises six distinct files, providing trait data, metadata, soil data, weather data, genotype data, and environmental data. The primary key employed for joining across the data sources, excluding genotype data, is &#x201c;Env,&#x201d; a combination of location and evaluation year. The genotype data was integrated with trait data using &#x201c;Hybrid&#x201d; as the key index.</p>
<sec id="s2_1_1">
<label>2.1.1</label>
<title>Metadata</title>
<p>The metadata includes details on the location, irrigation, planting date, specific issues encountered during the season, and agronomic management treatments, with the primary key being &#x201c;Env.&#x201d; The data covers locations from 21 states with various treatment types: standard, drought, irrigated, disease trial, early planting, late planting, late stressed, and dryland. Missing values for treatment were imputed using the mode of the column (standard treatment). Issues were manually classified into six categories based on raw comments: animal attack, data issues, drought, storm, no issues, and miscellaneous. The planting date, originally in date format, was converted to the day of the year format.</p>
</sec>
<sec id="s2_1_2">
<label>2.1.2</label>
<title>Weather data</title>
<p>Daily records of 16 distinct weather features were available from 2014 to 2021. Initially formatted longitudinally with the primary keys &#x201c;Env&#x201d; and date, the dataset was subsequently transformed into a wide format for modeling purposes. In this format, the primary key became &#x201c;Env,&#x201d; and the feature values for each day were transposed into columns, identified by names like &#x201c;FeatureName_DayOfYear.&#x201d; Furthermore, this wide dataset was subjected to additional preprocessing to aggregate the daily weather features into a weekly format.</p>
</sec>
<sec id="s2_1_3">
<label>2.1.3</label>
<title>Soil data</title>
<p>The soil data contained 23 fundamental soil information, with an absence of data for the year 2014. To tackle this issue, missing values for the respective features were imputed using the mean value from other years at the same location. Despite this imputation, certain locations still lacked soil features. To resolve this, a secondary imputation was performed, replacing missing soil values with the mean value of the corresponding state.</p>
</sec>
<sec id="s2_1_4">
<label>2.1.4</label>
<title>Environmental data</title>
<p>The variables in the environmental data are simulated outputs from crop simulation software named Agricultural Production Systems sIMulator (APSIM) developed by <xref ref-type="bibr" rid="B37">Lopez-Cruz et&#xa0;al. (2023)</xref>. The data set included various simulated soil features such as the water supply-demand ratio, extractable soil water ratio, water movement (upwards and downwards), nitrogen leaching as NO3, soil water content, and plant-available water. These features were recorded at 10 different depths and 9 phenological stages of the crop.</p>
<p>The data also included simulated phenological features such as grain yield, above-ground biomass, water table, and leaf area index at 9 phenological stages of the crop. However, some simulated variables had missing values for certain location-year combinations. These missing values were imputed using the mean value of the respective feature for the same location.</p>
</sec>
<sec id="s2_1_5">
<label>2.1.5</label>
<title>Genotype data</title>
<p>The initial genotype data encompassed information for 434,893 loci for 4928 hybrids. This data was preprocessed to structure the first column as the hybrid ID, while the following columns contained details on various loci. Out of these loci, the loci with missing values were dropped, which reduced the number to 45,846. It was observed that for those loci, more than 80% of hybrids had missing values. Hence, those loci were disregarded in future steps. The genotype data had values &#x201c;0/0&#x201d;, &#x201c;0/1&#x201d;, &#x201c;1/0&#x201d;, and &#x201c;1/1&#x201d; for the majority of the data. However, less than 5% of the entire data had some different values such as: &#x201c;2/0&#x201d;, &#x201c;0/2&#x201d;, &#x201c;2/1&#x201d;, &#x201c;1/2&#x201d;, &#x201c;2/2&#x201d;, &#x201c;3/0&#x201d;. A data encoding logic was applied where &#x201c;0/0&#x201d; was encoded as 0, &#x201c;0/1&#x201d; or &#x201c;1/0&#x201d; was encoded as 0.5, and &#x201c;1/1&#x201d; was encoded as 1, while any other values were encoded as 0.15. Encoding categorical values as numeric values is a well-established technique for representing DNA sequence data (<xref ref-type="bibr" rid="B38">Ma et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B47">Montesinos-L&#xf3;pez et&#xa0;al., 2018a</xref>; <xref ref-type="bibr" rid="B46">b</xref>; <xref ref-type="bibr" rid="B31">Khaki and Wang, 2019</xref>; <xref ref-type="bibr" rid="B25">Grinberg et&#xa0;al., 2020</xref>).</p>
<p>To reduce the dimensionality of genotype data, one strategy involves selecting loci with unique variations (<xref ref-type="bibr" rid="B45">Monroe et&#xa0;al., 2020</xref>). An empirical rule was applied to choose loci with diverse information. After encoding, loci where less than 4000 hybrids have a value of 0 and at least 1000 hybrids have a value of 0.5 and do not have 0.15 were selected. This selection reduced the number of loci to 4,227 while ensuring that loci do not have only zeros for all hybrids and identify loci where hybrids have different values. From these, 4,227 loci randomly, 300 were selected and used in the following analysis steps. Interestingly, the models indicated that performance remained consistent when using any randomly selected 300 loci from the full set of 4,227.</p>
</sec>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Data exploration</title>
<p>Data exploration was conducted to understand yield variations across different locations, such as states, by examining yield distribution and mean yield response. The dataset included yield values for various treatment types, including drought, irrigation, and late planting disease trials. The impact of these different treatment types on yield was analyzed and found that planting date had a significant impact on yield. Hence, the study investigated yield variations based on different planting dates.</p>
<p>Yield distribution across states: To understand yield distribution across states, the yield distribution for each state was visually inspected and compared. Additionally, new features were created to determine the mean yield for each hybrid and state combination, which were then used in building the yield prediction model.</p>
<p>Yield distribution across treatment types: Another focus was on understanding how yield varies across different treatment types. For example, to examine how yield distribution changes from standard conditions to drought conditions, the yield distributions for various treatment types were compared, along with the mean yield for each treatment type.</p>
<p>Impact of planting date on yield: The planting date is one of the vital decisions in plant management (<xref ref-type="bibr" rid="B18">Dobor et&#xa0;al., 2016</xref>). While optimum planting dates vary based on spatial location (<xref ref-type="bibr" rid="B64">Thorburn et&#xa0;al., 2017</xref>), and in some locations, late planting can reduce crop yield (<xref ref-type="bibr" rid="B66">Tsimba et&#xa0;al., 2013</xref>). In this research the impact of planting date was analyzed across 20 states and considering more than 4000 crop hybrids. The impact of planting date on yield was analyzed by the following five-step:</p>
<list list-type="simple">
<list-item>
<p>
<italic>Step 1:</italic> Compare the yield distribution for different planting dates across states.</p>
</list-item>
<list-item>
<p>
<italic>Step 2:</italic> Identify planting date and state combinations corresponding to lower or higher yield.</p>
</list-item>
<list-item>
<p>
<italic>Step 3:</italic> Identify hybrids with lower or higher yields.</p>
</list-item>
<list-item>
<p>
<italic>Step 4:</italic> Compare those hybrids&#x2019; standard planting time with planting time in other treatments.</p>
</list-item>
<list-item>
<p>
<italic>Step 5:</italic> Compare those hybrids&#x2019; yield distribution in standard planting time versus different planting time (early or late).</p>
</list-item>
</list>
<p>First, yield distributions for different planting dates were visualized and compared in terms of mean and standard deviation. The second step of the analysis revealed that the &#x201c;late planting&#x201d; treatment resulted in lower yields in Texas. In the third step, hybrids planted during the late-planting period were identified, and their planting dates in standard, drought, and irrigation treatments were compared in the later step. Finally, yield distributions at the hybrid level were compared for standard and late planting treatments to understand the impact of planting date on hybrids.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>High-yield hybrid identification</title>
<p>We developed a tool to identify hybrids with higher yields across various treatment types. The tool aids in selecting high-yield hybrids for specific scenarios. To build this tool, the yield distribution of each hybrid was analyzed in the first step. In the next step, an empirical classification logic was established using the maximum, minimum, median, and mean yields of each hybrid (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). In the last step, these classification rules were used to create a heatmap, where each box and its color represent a hybrid&#x2019;s yield performance for a treatment type. High-yield hybrids can be identified by visually locating those in the high-yield group across all treatments.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Empirical hybrid classification rule based on yield distribution.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g001.tif">
<alt-text content-type="machine-generated">Flowchart of hybrid yield distribution categorized by minimum, maximum, and median values. If maximum is less than 4.5 Mg/ha, yield is &#x201c;Extremely low.&#x201d; If median is less than 7 or mean less than 9, yield is &#x201c;Low.&#x201d; If minimum is less than 4 and maximum is greater than 12, yield is &#x201c;Wide range.&#x201d; If median is between 7 and 11, yield is &#x201c;Moderate.&#x201d; If minimum is greater than 10, yield is &#x201c;Extremely high.&#x201d; Otherwise, yield is &#x201c;High."</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Predictive modeling</title>
<p>A yield prediction model was developed by integrating genotype, environment, and management data. Various modeling approaches were explored, including linear models, tree-based models, and neural networks. To enhance model performance, new features were created before building the prediction model. After evaluating the performance of different models, an ensemble of tree-based models and neural networks demonstrated the best performance. The model was trained on data from the year 2014 to 2020, and 2021 served as an independent testing set.</p>
<p>This section outlines the methodology for the proposed prediction model across three subsections. Subsection 2.4.1 introduces additional features engineered to enhance model performance, derived from patterns and insights identified during the earlier exploratory analysis. Subsection 2.4.2 details the baseline models used for performance benchmarking. Subsection 2.4.3 provides a detailed explanation of the hybrid CNN-DNN + XGBoost model, describing how heterogeneous data sources (genotype, environment, and management) are processed and integrated through separate input channels to effectively capture complex interactions. The section systematically covers all steps leading up to prediction, including data processing, feature engineering, and model comparison.</p>
<sec id="s2_4_1">
<label>2.4.1</label>
<title>Feature engineering</title>
<sec id="s2_4_1_1">
<label>2.4.1.1</label>
<title>State and hybrid-level features</title>
<p>To account for how yield may vary by hybrid and state, features were generated for each hybrid-state combination by calculating the mean, maximum, and minimum yields using data up to 2020, the final year of the training set. For the test year 2021, there were many new hybrid-state combinations. Since this data was unavailable in the training set, missing values were imputed by averaging the mean and mode of the column grouped by state. These features were then incorporated into the prediction model.</p>
</sec>
<sec id="s2_4_1_2">
<label>2.4.1.2</label>
<title>Hybrid level features</title>
<p>A set of features was created at the hybrid level to help the model understand yield distribution variations by hybrid. For each hybrid, the mean, maximum, and minimum yields during training were calculated and used as features. Since some hybrids appeared only once or twice in the training data, their mean, maximum, and minimum yields did not adequately represent their yield distribution. To address this, two additional features were created to represent the mean yield of both hybrid parents.</p>
</sec>
<sec id="s2_4_1_3">
<label>2.4.1.3</label>
<title>Yield trend feature</title>
<p>The yield trend feature was generated at the state level due to incomplete yearly yield data for individual fields. A univariate linear regression model was applied to each state&#x2019;s yield values from the training years. The yield trend for year n was then calculated using the regression coefficients (<xref ref-type="disp-formula" rid="eq1">Equation 1</xref>).</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>d</mml:mi>
</mml:mstyle>
<mml:mo>&#xa0;</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
</mml:mstyle>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>d</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>s</mml:mi>
<mml:mi>n</mml:mi>
</mml:mstyle>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>a</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
</mml:mstyle>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>a</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>s</mml:mi>
</mml:mstyle>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>n</mml:mi>
</mml:mstyle>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where,</p>
<p>
<inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>yield&#xa0;trend</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>sn</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>yield&#xa0;trend&#xa0;for&#xa0;state&#xa0;s&#xa0;in&#xa0;year&#xa0;n</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mtext>a</mml:mtext>
<mml:mrow>
<mml:mtext>so</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>intercept&#xa0;for&#xa0;state&#xa0;s</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mtext>a</mml:mtext>
<mml:mrow>
<mml:mtext>s</mml:mtext>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>slope&#xa0;for&#xa0;state&#xa0;s</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec id="s2_4_2">
<label>2.4.2</label>
<title>Base learning models</title>
<sec id="s2_4_2_1">
<label>2.4.2.1</label>
<title>LASSO</title>
<p>Lasso is a linear regression variant with an L1 regularization feature in the model&#x2019;s loss function, which helps to build a model with only important features by assigning zero to less important ones, thus avoiding overfitting (<xref ref-type="bibr" rid="B65">Tibshiranit, 1996</xref>; <xref ref-type="bibr" rid="B29">James et&#xa0;al., 2013</xref>). In this research, Lasso was implemented using the scikit-learn package (<xref ref-type="bibr" rid="B8">Buitinck et&#xa0;al., 2013</xref>) in Python to make yield predictions. The optimal L1 regularization penalty was set to 0.05 based on a grid search.</p>
</sec>
<sec id="s2_4_2_2">
<label>2.4.2.2</label>
<title>Random Forest regressor</title>
<p>Random Forest (RF) is an ensemble tree-based model that creates multiple uncorrelated trees using a bootstrap resampling method (<xref ref-type="bibr" rid="B6">Breiman, 2001</xref>; <xref ref-type="bibr" rid="B13">Cutler et&#xa0;al., 2007</xref>). Each tree is constructed on a subsample of data and a subset of input features (<xref ref-type="bibr" rid="B7">Brown, 2017</xref>), continuing this process until all trees are formed (<xref ref-type="bibr" rid="B13">Cutler et&#xa0;al., 2007</xref>). For regression, the final prediction is the average of these trees. This research implemented the RF model using the RandomForestRegressor from the scikit-learn package (<xref ref-type="bibr" rid="B8">Buitinck et&#xa0;al., 2013</xref>). Hyperparameters were selected through a grid search, while default values were used for other parameters (<xref ref-type="table" rid="T1">
<bold>Table 1</bold>
</xref>).</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Optimum hyperparater values used for RF Regressor.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Hyperparameter</th>
<th valign="top" align="left">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Maximum depth:</td>
<td valign="top" align="left">15</td>
</tr>
<tr>
<td valign="top" align="left">Maximum number of features:</td>
<td valign="top" align="left">Square-root of total number of features</td>
</tr>
<tr>
<td valign="top" align="left">Minimum sample required to split:</td>
<td valign="top" align="left">15</td>
</tr>
<tr>
<td valign="top" align="left">Number of trees:</td>
<td valign="top" align="left">400</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_4_2_3">
<label>2.4.2.3</label>
<title>XGBoost regressor</title>
<p>The gradient-boosting tree-based model, XGBoost, was employed for yield prediction. This model learns sequentially from weak learners, with trees generated using the &#x201c;exact&#x201d; method (<xref ref-type="bibr" rid="B9">Chen and Guestrin, 2016</xref>). The final prediction is made by aggregating all weak base learners. Optimal hyperparameters for the best-performing model were found using a grid search method (<xref ref-type="table" rid="T2"><bold>Table 2</bold></xref>).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Optimum hyperparater values used for XGBoost Regressor.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Hyperparameter</th>
<th valign="top" align="left">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Maximum depth:</td>
<td valign="top" align="left">5</td>
</tr>
<tr>
<td valign="top" align="left">Eta (step size shrinkage):</td>
<td valign="top" align="left">0.05</td>
</tr>
<tr>
<td valign="top" align="left">Sub sample:</td>
<td valign="top" align="left">0.75</td>
</tr>
<tr>
<td valign="top" align="left">Number of trees:</td>
<td valign="top" align="left">4000</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2_4_2_4">
<label>2.4.2.4</label>
<title>Single modal CNN-DNN</title>
<p>A single-modal CNN-DNN model was designed to predict yield, combining all numerical data sources and encoding categorical features (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>). These inputs were passed through an 8-layer CNN-DNN model. The input layer was followed by two 1D CNN layers with 32 and 16 filters, respectively, both having a kernel size of 5 and an average pooling layer. This was connected to two more 1D CNN layers, followed by a flattening layer and three fully connected layers. The final fully connected layer served as the output layer for yield prediction (<xref ref-type="table" rid="T3"><bold>Table 3</bold></xref>).</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Single modal CNN-DNN architecture with four 1D CNN layers followed by two fully connected dense layers.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g002.tif">
<alt-text content-type="machine-generated">Flowchart of a neural network architecture for yield prediction.  It includes four 1D CNN layers with filter sizes of thirty-two, sixteen, and eight, and kernel sizes of five and three, respectively, with ELU activation. An average pool layer with a pool size and stride of two follows the second CNN. The fourth CNN is followed by flattening, two dense layers with ten and six units, ELU activation, and finally, a yield prediction output.</alt-text>
</graphic>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Hyperparater values used for Single modal CNN-DNN.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Hyperparameter</th>
<th valign="top" align="left">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Maximum depth:</td>
<td valign="top" align="left">3</td>
</tr>
<tr>
<td valign="top" align="left">Eta (step size shrinkage):</td>
<td valign="top" align="left">0.01</td>
</tr>
<tr>
<td valign="top" align="left">Learning rate:</td>
<td valign="top" align="left">0.01</td>
</tr>
<tr>
<td valign="top" align="left">Sub sample:</td>
<td valign="top" align="left">0.15</td>
</tr>
<tr>
<td valign="top" align="left">Colsample_bytree:</td>
<td valign="top" align="left">0.15</td>
</tr>
<tr>
<td valign="top" align="left">Number of trees:</td>
<td valign="top" align="left">400</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s2_4_3">
<label>2.4.3</label>
<title>Proposed CNN-DNN with XGBoost model</title>
<p>The proposed CNN-DNN with the XGBoost model is an ensemble model that combines CNN-DNN and XGBoost models. The XGBoost model was trained on metadata and features representing mean yield at the state level, hybrid level, and hybrid-state yield combinations. Meanwhile, the CNN-DNN model was trained on weather, soil, environmental (from APSIM), genotype data, and metadata.</p>
<p>For the weather block, the 16 district weather features from weeks 1 to 48 were converted into a 3D array. The input layer for the weather block had a shape of (n&#xd7;48&#xd7;16) to maintain the time sequence, which then passed through two 2D CNN layers, followed by average pooling, flattening, and dense layers (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>CNN block architecture. Two CNN layers are followed by an average pooling, flattening, and fully connected layer.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g003.tif">
<alt-text content-type="machine-generated">Diagram of a neural network architecture. It consists of two CNN layers shown as stacked squares, followed by an average pooling layer. This is connected to a flatten layer, and concludes with a dense layer.</alt-text>
</graphic>
</fig>
<p>The 23 soil features and 300 genotype features lacked any specific order, so the soil and genotype blocks used 1D-CNN layers. The metadata included categorical features, such as &#x201c;field location&#x201d; and &#x201c;crop rotation,&#x201d; which were incorporated into the CNN-DNN model and processed through two embedding layers. These embedding layers, along with other numerical features, were then passed through two 1D-CNN layers followed by a flattening layer (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>CNN block with embedding layers. The embedding layers and numeric metadata are concatenated, which is then followed by two 1D CNN layers and a flattened layer.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g004.tif">
<alt-text content-type="machine-generated">Diagram showing a data processing pipeline. Three inputs, &#x201c;Location,&#x201d; &#x201c;Crop rotation,&#x201d; passing through an &#x201c;Embedding&#x201d; and &#x201c;Numeric metadata,&#x201d;. These are concatenated and flow into two sequential &#x201c;CNN&#x201d; (Convolutional Neural Networks) layers, followed by a &#x201c;Flatten&#x201d; operation.</alt-text>
</graphic>
</fig>
<p>Environmental features included soil-related data at 10 depths and 9 phenological stages, except for the Flow feature, which had values at 9 depths and 9 phenological stages. These soil features were reshaped to (n&#xd7;10&#xd7;9), while the Flow feature values were reshaped to (n&#xd7;9&#xd7;9). Each feature passed through two 2D CNN layers, followed by average pooling, flattening, and dense layers. The environmental soil block comprised 7 separate CNN blocks, which were concatenated to form the output of the environmental soil block (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Environmental soil feature block, where each soil feature at different soil depths and phenological stages passes through 2D CNN layers. At the end, outputs from the soil features are concatenated.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g005.tif">
<alt-text content-type="machine-generated">Diagram showing a multi-input convolutional neural network (CNN) architecture. It has six input branches labeled SRD, ESW, Flow, Flow NO3, Flux, PAW mm, and SW mm. Each branch passes through a CNN, average pooling, flattening, and dense layers. All branches are concatenated into a single output layer.</alt-text>
</graphic>
</fig>
<p>The CNN-DNN model includes six separate CNN blocks for different types of input data, with five blocks having similar architectures. Each block contains two CNN layers followed by average pooling, flattening, and dense layers (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>). Depending on the input features, some blocks used 1D CNN layers, while others used 2D CNN layers (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>).</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Proposed CNN-DNN framework with XGBoost model. The weather data, environmental soil data (from APSIM), and environmental phenological data (from APSIM) flow through separate 1D CNN blocks, while the soil data and genotype data pass through 1D CNN blocks. The metadata passes through the 1D CNN layer with embedding layers. The output of all the blocks is concatenated, followed by the DNN layer and the output layer. This yield prediction is ensembled with the yield prediction from XGBoost, trained on hybrid and state-level yield features only.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g006.tif">
<alt-text content-type="machine-generated">Diagram illustrating the ensemble of CNN-DNN and XGBoost models for yield prediction. The top model uses a CNN-DNN architecture, processing various data types like weather, soil, and genotype through different CNN layers, then concatenates them into a DNN for prediction. The bottom model involves XGBoost, starting with derived yield features, applying over-sampling, proceeding with the XGBoost model, and ending with yield prediction. Both models lead to a final yield prediction.</alt-text>
</graphic>
</fig>
<p>The output of all six blocks was concatenated and passed through two fully connected layers with L1-regularization to avoid over-fitting on training data (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). The number of kernels, filters, and activation functions used in each layer is detailed in <xref ref-type="supplementary-material" rid="SM1">
<bold>Table 1</bold>
</xref> of <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>
</p>
<p>The proposed architecture employed the exponential linear unit (ELU) activation function for all layers except the output layer. The choice of ELU over the rectified linear unit (ReLU) was based on its advantage in allowing negative values, thereby causing the unit activations to approach zero with reduced computational complexity (<xref ref-type="bibr" rid="B10">Clevert et&#xa0;al., 2015</xref>). Following is the mathematical formula used in ELU activation</p>
<p>ELU with <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mtext>&#x3b1;</mml:mtext>
<mml:mo>&gt;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<disp-formula>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>g</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>x</mml:mi>
</mml:mstyle>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>x</mml:mi>
</mml:mstyle>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
</mml:mstyle>
<mml:mo>&#xa0;</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>x</mml:mi>
</mml:mstyle>
<mml:mo>&gt;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>g</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>x</mml:mi>
</mml:mstyle>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>&#x3b1;</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>x</mml:mi>
</mml:mstyle>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
</mml:mstyle>
<mml:mo>&#xa0;</mml:mo>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>x</mml:mi>
</mml:mstyle>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<p>XGBoost with oversampling: To improve predictability for drought, disease trials, or late planting time, oversampling was applied to the training set because there were fewer observations for these treatments compared to the standard treatment. For each treatment type, excluding standard and late planting, 4,000 samples were randomly selected with replacement. As the training data had no treatment corresponding to the late planting scenario, an additional 1,000 samples were selected from each treatment type, renaming the treatment type to late planting, resulting in an additional 25,000 observations along with the real data. This oversampled data was used to build the prediction model, using only metadata, hybrid-level, and state-hybrid-level yield features (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). The XGBoost model was employed for prediction with hyperparameters selected from a grid search approach.</p>
<p>The predictions from the XGBoost model and the CNN-DNN model were combined using a weighted average. To determine the optimal weights for combining CNN-DNN and XGBoost, we performed a grid search by varying the weights in 0.1 increments. The optimal weights were assigned as 0.1 to the XGBoost model predictions and 0.9 to the CNN-DNN model predictions.</p>
<p>All models, baseline models, and proposed CNN-DNN with and without XGBoost, were evaluated in terms of multiple metrics. Those are root mean squared error (RMSE), relative root mean squared error (RRMSE), mean absolute percentage error (MAPE) and Pearson correlation. The model with the lowest RMSE, RRMSE, MAPE and higher Pearson correlation was selected for prediction.</p>
<p>RMSE is calculated utilizing <xref ref-type="disp-formula" rid="eq4">Equation 2</xref>.</p>
<disp-formula id="eq4">
<label>(2)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mtext>RMSE</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>i</mml:mi>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>n</mml:mi>
</mml:mstyle>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>y</mml:mi>
</mml:mstyle>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>i</mml:mi>
</mml:mstyle>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:mstyle>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>i</mml:mi>
</mml:mstyle>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>n</mml:mi>
</mml:mstyle>
</mml:mfrac>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<p>RRMSE is calculated using <xref ref-type="disp-formula" rid="eq5">Equation 3</xref>.</p>
<disp-formula id="eq5">
<label>(3)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mtext>RRMSE</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>RMSE</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>MAPE is calculated by <xref ref-type="disp-formula" rid="eq6">Equation 4</xref>.</p>
<disp-formula id="eq6">
<label>(4)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mtext>MAPE</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>i</mml:mi>
</mml:mstyle>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>n</mml:mi>
</mml:mstyle>
</mml:msubsup>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mtext>y</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Pearson correlation is calculated through <xref ref-type="disp-formula" rid="eq7">Equation 5</xref>.</p>
<disp-formula id="eq7">
<label>(5)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mtext>Correlation</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where,</p>
<p>
<inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mtext>y</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mtext>&#xa0;is&#xa0;the&#xa0;ith&#xa0;obsevarion&#xa0;of&#xa0;the&#xa0;response&#xa0;variable</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mtext>&#xa0;is&#xa0;the&#xa0;ith&#xa0;prediction&#xa0;of&#xa0;the&#xa0;response&#xa0;variable</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">m</mml:mtext>
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
</mml:msub>
<mml:mtext>&#xa0;mean&#xa0;of&#xa0;the&#xa0;obsevred&#xa0;response&#xa0;variable</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mtext mathvariant="bold-italic">m</mml:mtext>
<mml:mover accent="true">
<mml:mtext mathvariant="bold-italic">y</mml:mtext>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:msub>
<mml:mtext>&#xa0;mean&#xa0;of&#xa0;the&#xa0;prediction&#xa0;of&#xa0;&#xa0;response&#xa0;variable</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and analyses</title>
<p>This section presents the findings from analyzing factors impacting crop yields, including location, hybrids, and management decisions. Additionally, it includes the hybrid classification results using the empirical rule-based tool for identifying high-yield hybrids in different conditions (i.e., drought, disease trail). Finally, the results and performance evaluations of the predictive models are provided.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Exploratory analysis</title>
<p>Exploratory analysis was conducted to assess the impact of G&#xd7;E&#xd7;M factors on crop yield and identify those with the greatest influence. The analysis specifically examined the effects of location, management practices, and hybrid type on crop yield.</p>
<p>
<italic>Yield distribution across states:</italic> A comparison of mean yield and the number of unique hybrids planted at the state level revealed that areas with lower mean yields tend to have fewer hybrid varieties. For instance, states like Colorado, South Dakota, Arkansas, and South Carolina have 235 to 958 hybrid varieties, with mean yield ranging from 5.12 to 6.92 Mg/ha. In contrast, states such as Iowa, Illinois, and Indiana have mean yield ranging from 10.87 to 11.30 Mg/ha, with more than 1,500 different hybrids planted. This suggests that states with a higher number of different hybrids tend to have higher average yield (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>).</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Mean yield variation in different states along with different numbers of hybrids planted. Yellow circles represent mean yield, while a larger circle corresponds to higher yield. The colors represent the number of unique hybrids planted, where sky blue is for lower hybrid variety.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g007.tif">
<alt-text content-type="machine-generated">Map of the United States displaying mean yield and number of unique hybrids by state. Yellow circles indicate mean yield in megagrams per hectare, with larger circles representing higher yields. States are color-coded based on the number of unique hybrids: light blue indicates two hundred thirty-five to nine hundred fifty-eight, blue for nine hundred fifty-nine to one thousand five hundred, and increasing in shades up to pink for two thousand nine hundred ninety-two to three thousand eight hundred twenty-nine. A compass and scale in miles are included.</alt-text>
</graphic>
</fig>
<p>Further analysis of yield distribution at the state level revealed significant variation among states. For instance, Iowa, Illinois, Indiana, and Missouri have unimodal yield distributions, though the mode varies. Conversely, other states, Arkansas and Kansas display a bimodal distribution, indicating that yield distribution varies based on demographic locations (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material Figure&#xa0;1</bold>
</xref>).</p>
<p>Further analysis compared hybrids planted across multiple states (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>). The yield distribution comparison revealed that a hybrid with a high yield in one state might have a lower yield in another. For example, the hybrid &#x201c;2369/LH123HT&#x201d; has its 25th percentile yield above 10 Mg/ha in states like Delaware, Iowa, Illinois, Indiana, Michigan, Ohio, Wisconsin, and Georgia. However, in Colorado, South Carolina, and Arkansas, its 75th percentile yield is less than 8 Mg/ha. Conversely, hybrids like &#x201c;B73/MO17&#x201d; and &#x201c;B73/PHN82&#x201d; have higher mean yields in Colorado compared to &#x201c;2369/LH123HT.&#x201d; This indicates that a hybrid&#x2019;s high yield in one location does not guarantee similar performance elsewhere, highlighting the crucial role of hybrid-environment interactions in crop yield.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Variability in crop yield distribution across locations for the same hybrid.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g008.tif">
<alt-text content-type="machine-generated">Boxplots showing yield (Mg/Ha) for various corn hybrids across different states. Each plot represents a different hybrid, with states on the x-axis and yield on the y-axis. Variability in yield is visible across different states for each hybrid.</alt-text>
</graphic>
</fig>
<p>
<italic>Yield distribution across treatment types:</italic> The yield distribution varies across different treatments (<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>). Standard treatment has the highest mean yield (~10 Mg/ha) and higher values at the 25th and 75th percentiles. Treatments such as irrigation, disease trials, dry land, and late stress show lower mean yield values than standard treatment, indicating reduced yield under these agronomic conditions. The yield distribution for drought conditions is bimodal, with modes at 6 Mg/ha and 12 Mg/ha, indicating that yield in drought can be moderate or lower. While, in Texas, the late planting treatment caused the lowest yield, highlighting the significant impact of planting decisions on yield. This observation prompted a detailed analysis of the planting date&#x2019;s effect on crop yield.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Yield distribution for different treatment types. The standard condition has a distribution with the highest mean, and late planting treatment has the lowest mean.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g009.tif">
<alt-text content-type="machine-generated">Violin plot showing crop yield in megagrams per hectare for different treatments: Standard, Irrigated, Disease Trial, Drought, DryLand, Late Stressed, and Late Planting. Each treatment displays variability and distribution in yield.</alt-text>
</graphic>
</fig>
<p>
<italic>Impact of planting date on yield:</italic> The five steps mentioned in section 2.2 to analyze the impact of planting date were applied to the entire data. The analysis starts with identifying the difference in yield distribution for different planting dates and expanding to yield distribution for hybrid and planting date combinations.</p>
<p>Step 1- Yield distribution comparison for different planting dates across states:</p>
<p>The impact of planting dates on yield varies significantly across states (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material Figure&#xa0;2</bold>
</xref>). For instance, in Georgia, planting in mid-May (day 132 of the year) results in a higher yield distribution mean, while in South Carolina, planting after April (day 120) yields a lower mean. States in close geographical proximity often have similar planting dates. For example, Iowa, Illinois, and Nebraska typically plant from late April (day 114) to the end of May (day 150). Conversely, in warmer states like Texas, planting dates range from early March to early May (day 60 to day 128). Therefore, what is considered early planting in colder regions might be late for warmer locations, indicating that the impact of planting dates on crop yield varies by location.</p>
<p>Steps 2 &amp; 3- Identify planting date, state combinations, and hybrids corresponding to lower yield:</p>
<p>In Texas, planting around mid-April (day 99) resulted in lower yield distribution and was categorized as a late planting condition. Further analysis was conducted at the hybrid level, selecting 21 hybrids planted late in Texas with yield data across all treatments. These hybrid IDs are encoded in letters and, in the rest of the paper, will be presented using these letters as given in <xref ref-type="supplementary-material" rid="SM1">
<bold>Table&#xa0;2</bold>
</xref> of <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Materials</bold>
</xref>.</p>
<p>Step 4 &#x2013; Comparing selected hybrids&#x2019; planting time in a standard scenario with other treatments:</p>
<p>The standard treatment planting date distribution revealed that these hybrids are typically planted in Texas in the third week of March (day 80). For colder locations, the planting dates for these hybrids ranged from mid-April to the end of May (days 107 to 150) (<xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>). These dates were considered standard based on the local weather conditions. The planting periods for these hybrids were also examined for other treatments: early May to late May (days 125 to 150) for drought treatment, the last two weeks of May (days 140 to 154) for disease trials, and mid-March to mid-May for dryland (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material Figure&#xa0;3</bold>
</xref>). These planting windows for other treatments in various locations align with the standard planting times. The mid-April planting date in Texas is outside the standard planting window for these hybrids for that location.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>The boxplot of planting dates in different states in standard condition for 21 hybrids planted during the late planting scenario in Texas. The planting dates vary across states.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g010.tif">
<alt-text content-type="machine-generated">Box plots show planting dates for different hybrids across 18 states: TX, NY, IA, IN, GE, GA, WI, MO, OH, KS, AR, SC, NE, NC, IL, MI, MN, and DE. Each plot indicates data distribution, variability, and outliers for hybrids labeled A to U, with planting dates ranging from 80 to 160 days.</alt-text>
</graphic>
</fig>
<p>Step 5- Compare hybrids&#x2019; yield distribution in standard planting time versus late planting time:</p>
<p>Comparing these hybrids&#x2019; yield distribution in standard treatment and late planting time revealed that yields were lower when planting dates differed from the standard time (<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>). All 21 hybrids exhibit substantially lower yield distribution when planted late in the season.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Comparison of yield distribution for hybrids in standard and late planting times. Late planting results in lower yield compared to standard planting time.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g011.tif">
<alt-text content-type="machine-generated">Box plot chart comparing yield of different hybrids under standard (blue) and late planting (orange) conditions. Hybrids labeled A to U along the x-axis and yield in Mg/Ha on the y-axis. Each box represents yield distribution, with outliers marked as diamonds.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>High-yield hybrid identification tool</title>
<p>The visual tool for identifying hybrids is color-coded based on their yield categories: extremely low, low, moderate, wide range, high, and extremely high. These classifications are derived from the yield distribution rules outlined in section 2.3 that use empirical classification based on yield distribution (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). A prototype of the visual tool includes 21 hybrids commonly found across various treatment types. Red indicates hybrids with extremely low yields in all treatment combinations, while blue represents hybrids with extremely high yields (<xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>).</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Empirical rule-based tool to identify hybrids with high yield for different treatments. Each column represents a different treatment, and each row represents a hybrid. Different color corresponds to yield class.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g012.tif">
<alt-text content-type="machine-generated">Heatmap displaying performance of different hybrids across various trial conditions. Rows represent hybrids labeled A to U, and columns represent conditions such as Disease trial, Drought, Dryland, and others. Color scale ranges from extremely low (red) to extremely high (blue) with values like high (green) and moderate (yellow).</alt-text>
</graphic>
</fig>
<p>This tool aids in selecting high-yield hybrids, particularly for locations prone to drought, dryland, and disease. For example, hybrids such as TX7777/LH195 (<bold>U</bold>), PHW52/PHN82 (<bold>T</bold>), F42/OH43 (<bold>O</bold>), F42/H95 (<bold>M</bold>), CG444/CGR01 (<bold>L</bold>), B37/MO17 (<bold>H</bold>), and B37/OH43 (<bold>G</bold>), which correspond to high yield categories in both standard and drought conditions, are ideal choices. It helps identify hybrids with moderate to high yields across most treatment types. For instance, the hybrid TX777/LH195(A) is suitable for scenarios involving disease attacks, drought, and standard conditions. The tool can be modified to suggest hybrids suited for specific treatment types for a broader selection.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Yield prediction and analysis</title>
<p>The prediction model was trained on data from 2014 to 2020, including yield scenarios from 21 states. The training data had a mean yield of 9.44 Mg/ha, with a standard deviation of 2.99 Mg/ha, and ranged from a minimum of 0.5 Mg/ha to a maximum of 23.27 Mg/ha. This demonstrates a wide range of yield values across different states in the training data. The prediction models were evaluated using yield values from 2021, which had a mean of 10.04 Mg/ha. The yield distribution for the test year differed from the training years, with a standard deviation of 2.78 Mg/ha and yields ranging from 0.58 Mg/ha to 18.77 Mg/ha.</p>
<p>
<xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref> compares the yield distributions across training and testing sets for various U.S. states, including Delaware (DE), Georgia (GA), Iowa (IA), Illinois (IL), Indiana (IN), Nebraska (NE), New York (NY), Texas (TX) and Wisconsin (WI). The analysis indicated significant discrepancies between the yields in training and testing periods at the state level. For example, in Iowa, the training set&#x2019;s average yield was 11.5 Mg/ha, which decreased to 10.12 Mg/ha during the test year. Texas exhibited a reduction from 8.4 Mg/ha in the training set to 6.8 Mg/ha in the test year, with the test year also showing a bimodal distribution. Conversely, in New York and Nebraska, the test sets recorded higher mean yields (NY: 11 Mg/ha; NE: 10.4 Mg/ha) compared to the training sets (NY: 9.5 Mg/ha; NE: 7.8 Mg/ha), highlighting regional variations in yield consistency between the periods under review.</p>
<fig id="f13" position="float">
<label>Figure&#xa0;13</label>
<caption>
<p>Comparison of yield distribution for training (2014-2020) and test (2021) year.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g013.tif">
<alt-text content-type="machine-generated">Nine histograms show yield distribution for training and test data across different states: Delaware, Georgia, Iowa, Illinois, Indiana, Nebraska, New York, Texas, and Wisconsin. Each histogram features blue bars for training data and orange bars for test data, with yield measured in megagrams per hectare on the x-axis and count on the y-axis.</alt-text>
</graphic>
</fig>
<p>All six models were trained on the training data and evaluated using the test data in terms of RMSE, RRMSE, MAPE, and Pearson Correlation (<xref ref-type="fig" rid="f14">
<bold>Figure&#xa0;14</bold>
</xref>; <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material Table&#xa0;3</bold>
</xref>). Comparing all metrics revealed that Lasso had the lowest performance (RMSE: 3.31 Mg/ha; RRMSE: 33%; MAPE: 22%; Correlation: 0.28), followed by XGBoost trained on the entire data (RMSE of 3.14 Mg/ha, RRMSE of 31%, MAPE 25%, Correlation: 0.32), the simple CNN model (RMSE: 2.85 Mg/ha; RRMSE: 28%; MAPE: 23%; Correlation: 0.41), and the RF model (RMSE: 2.82 Mg/ha; RRMSE: 28%; MAPE: 22%; Correlation: 0.39), The proposed CNN-DNN had an RMSE of 2.46 Mg/ha and the CNN-DNN model ensembled with XGBoost (trained on yield features) had an RMSE of 2.45, while other matrics having similar values ((RRMSE: 24%; MAPE: 19%; Correlation: 0.51) (<xref ref-type="fig" rid="f14">
<bold>Figure&#xa0;14</bold>
</xref>). This indicates that the proposed CNN-DNN and CNN-DNN ensembled with XGBoost outperform other prediction models. It also highlights that due to the intricate relationship between genotype, environment, and management, a simple CNN with all these inputs fed together has lower predictability. The benefit of using modular blocks for different types of inputs is evident.</p>
<fig id="f14" position="float">
<label>Figure&#xa0;14</label>
<caption>
<p>Comparison of prediction model performances. The proposed CNN-DNN model and base models are evaluated in terms of <bold>(a)</bold> RMSE, <bold>(b)</bold> RRMSE, <bold>(c)</bold> MAPE, and <bold>(d)</bold> Pearson Correlation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g014.tif">
<alt-text content-type="machine-generated">Four bar charts comparing different models: (a) RMSE in Mg/ha, (b) RRMSE in percentage, (c) MAPE in percentage, and (d) Pearson Correlation. Models include CNN-DNN, CNN-DNN with XGBoost, XGBoost, RF, LASSO, and Simple CNN, with performance indicated by varying bar heights.</alt-text>
</graphic>
</fig>
<p>The dataset utilized in this study originates from the 2022 Genomes to Fields (G2F) competition (<xref ref-type="bibr" rid="B36">Lima et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B63">The Genomes To Fields Initiative, 2023</xref>). A comparative evaluation of the proposed CNN-DNN and CNN-DNN combined with XGBoost models against other submissions using the same dataset indicates that our model achieved an RMSE of 2.46 Mg/ha. This result falls within the range of the top 10 highest-performing models (<xref ref-type="bibr" rid="B69">Washburn et&#xa0;al., 2024</xref>), demonstrating that the proposed approach offers competitive predictive performance while additionally providing enhanced interpretability.</p>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>Uncovering the interactions between G&#xd7;E&#xd7;M factors and crop yield can enhance productivity. In this study, an exploratory analysis was conducted to identify key factors influencing crop yield under various scenarios, such as drought, standard conditions, disease trials, and irrigation. Results showed significant yield variation based on hybrid-location combinations, and planting dates were found to impact yield. Additionally, an empirical rule-based tool was developed for high-yield hybrid classification across different scenarios. Finally, a multimodal CNN-DNN model, ensembled with XGBoost, was created to predict yield considering G&#xd7;E&#xd7;M interactions. This section summarizes the key findings of this study.</p>
<p>Selecting informative loci from genotype data is crucial for yield prediction model performance. Using the entire genotype dataset can lead to a curse of dimensionality, so choosing an appropriate subset is essential. Iterative analysis revealed that selecting around 300 loci improves prediction accuracy, given the training data size. Systematic selection of these loci, as opposed to random selection, enhances prediction because random selection may include less informative loci. The proposed method for locus selection demonstrated better yield prediction results.</p>
<p>In addition to using appropriate feature selection methods, the imputation technique is important, particularly when dealing with datasets with many missing values. Given the locational dependency of the observations, location-based imputation significantly aided the prediction model&#x2019;s learning. However, observations with numerous missing values were excluded to prevent the model from learning from synthetic data. Deciding the extent of imputation and which observations to exclude was a key aspect of the data preprocessing stage.</p>
<p>The analysis of model architecture showed that a multimodal CNN-DNN model outperforms a single-modal CNN-DNN model in terms of predictability. In single-modal models, temporal and spatial dependencies are often lost when using various data sources for prediction. Notably, for weather data, a 2D CNN block was more effective than a 1D CNN or LSTM block in feature extraction and model prediction. Converting the 16 distinct weather features from week 1 to week 48 into a 3D array, with the y-axis representing the week number and the z-axis representing each weather feature, preserved temporal dependencies and improved the model&#x2019;s performance.</p>
<p>Exploring the model prediction results, it was observed that the mean yield for state-hybrid combinations is well correlated with the observed yield for combinations that appeared at least twice during the training data (Pearson correlation: 0.78 for training data; 0.35 for test data). However, for new hybrid-state combinations, this feature had less predictability, indicating that crop yield can vary widely due to G&#xd7;E interactions.</p>
<p>An evaluation of model performance across various treatments revealed that the model had the best predictability for irrigation (RMSE 2.31 Mg/ha) and standard treatment (RMSE 2.36 Mg/ha), followed by dryland (RMSE 2.48 Mg/ha) and drought (RMSE 2.85 Mg/ha). The highest error was observed in the late planting scenario (RMSE 3.91 Mg/ha), likely due to the training data containing only synthetic observations for this scenario. This suggests that the model performs better for treatments or scenarios included in the training data; however, it has limitations in predicting yield for new management practices it has not encountered. This limitation could be addressed by expanding the training dataset to include a broader range of management practices. Additionally, incorporating satellite data to capture crop growth stages may improve the model&#x2019;s ability to predict outcomes under extreme conditions.</p>
<p>Examining the observed and predicted yield distributions for various scenarios revealed that the distributions for the standard treatment were nearly identical (<xref ref-type="fig" rid="f15">
<bold>Figure&#xa0;15</bold>
</xref>). For drought, the observed yield distribution was higher (25th percentile at 11 Mg/ha, mean at 12 Mg/ha, and 75th percentile at 14.5 Mg/ha) compared to the predicted distribution (25th percentile at 9 Mg/ha, mean at 9.5 Mg/ha, and 75th percentile at 10 Mg/ha). Conversely, for late planting, the predicted yield distribution (25th percentile at 5 Mg/ha, mean at 5.5 Mg/ha, and 75th percentile at 6 Mg/ha) was higher than the observed distribution (25th percentile at 2 Mg/ha, mean at 2.5 Mg/ha, and 75th percentile at 4 Mg/ha). This discrepancy is due to the lack of real training data for late planting and the presence of both high and low-yield scenarios for drought in the training data. For other treatments, the predicted and observed yield distributions overlap.</p>
<fig id="f15" position="float">
<label>Figure&#xa0;15</label>
<caption>
<p>Evaluation of model performance for different treatment types.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1537990-g015.tif">
<alt-text content-type="machine-generated">Box plot comparing observed and predicted crop yields under different treatments: Drought, Standard, DryLand, Irrigated, and Late Planting. Yields range from 0 to 17.5. Observed data is in blue and predicted data in orange.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusion</title>
<p>This research investigated the factors influencing agricultural productivity, proposed a hybrid selection tool to improve yields under extreme weather conditions or specific locations, and developed a yield prediction model. The exploratory data analysis highlighted the variations in yield related to genotype, environment, and management practices. Particularly, yield varied significantly based on location and hybrid interaction, with stable historical data providing good predictability for future outcomes. A prediction model incorporating genotype, environment, and management interactions (G x E x M) was developed using a multimodal CNN-DNN model. This model demonstrated its effectiveness in predicting field-level productivity across various management practices, weather conditions, locations, and hybrids. Future research could focus on estimating other phenotypic traits at different crop cycle phases and incorporating spatial properties into the hybrid identification tool for various scenarios. To better&#xa0;address unforeseen conditions, in addition to increasing training data, crop simulation studies grounded in plant science&#xa0;and physiology could also be incorporated into the prediction framework.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <uri xlink:href="https://datacommons.cyverse.org/browse/iplant/home/shared/commons_repo/curated/GenomesToFields_G2F_data_2022">https://datacommons.cyverse.org/browse/iplant/home/shared/commons_repo/curated/GenomesToFields_G2F_data_2022</uri>.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>SS: Conceptualization, Data curation, Formal analysis, Methodology, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. ZK: Methodology, Validation, Writing &#x2013; review &amp; editing. LW: Methodology, Supervision, Validation, Writing &#x2013; review &amp; editing. GH: Funding acquisition, Methodology, Supervision, Validation, Writing &#x2013; review &amp; editing, Project administration.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was partially supported by NSF and USDA (#1830478 and #2021-67021-35329) and the Plant Sciences Institute at Iowa State University.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We want to acknowledge The Genomes To Fields (G2F) Initiative for sharing enormous datasets to research community.</p>
</ack>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2025.1537990/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2025.1537990/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
<supplementary-material xlink:href="Table1.docx" id="SM2" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anapalli</surname> <given-names>S. S.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Nielsen</surname> <given-names>D. C.</given-names>
</name>
<name>
<surname>Vigil</surname> <given-names>M. F.</given-names>
</name>
<name>
<surname>Ahuja</surname> <given-names>L. R.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Simulating planting date effects on corn production using RZWQM and CERES-maize models</article-title>. <source>Agron. J.</source> <volume>97</volume>, <fpage>58</fpage>&#x2013;<lpage>71</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/agronj2005.0058</pub-id>
</citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bali</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Singla</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Emerging trends in machine learning to predict crop yield and study its influential factors: A survey</article-title>. <source>Arch. Comput. Methods Eng.</source> <volume>29</volume>, <fpage>95</fpage>&#x2013;<lpage>112</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/S11831-021-09569-8</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baltrusaitis</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ahuja</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Morency</surname> <given-names>L. P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Multimodal machine learning: A survey and taxonomy</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>41</volume>, <fpage>423</fpage>&#x2013;<lpage>443</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2018.2798607</pub-id>, PMID: <pub-id pub-id-type="pmid">29994351</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beres</surname> <given-names>B. L.</given-names>
</name>
<name>
<surname>Hatfield</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Kirkegaard</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Eigenbrode</surname> <given-names>S. D.</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>W. L.</given-names>
</name>
<name>
<surname>Lollato</surname> <given-names>R. P.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Toward a better understanding of genotype &#xd7; Environment &#xd7; Management interactions&#x2014;A global wheat initiative agronomic research strategy</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/FPLS.2020.00828/BIBTEX</pub-id>, PMID: <pub-id pub-id-type="pmid">32612624</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bollero</surname> <given-names>G. A.</given-names>
</name>
<name>
<surname>Bullock</surname> <given-names>D. G.</given-names>
</name>
<name>
<surname>Hollinger</surname> <given-names>S. E.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Soil temperature and planting date effects on corn yield, leaf area, and plant development</article-title>. <source>Agron. J.</source> <volume>88</volume>, <page-range>385&#x2013;390</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/agronj1996.00021962008800030005x</pub-id>
</citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Brown</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Ensemble Learning</source> (<publisher-loc>Boston, MA</publisher-loc>: <publisher-name>Springer</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-1-4899-7687-1_252</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Buitinck</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Louppe</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Blondel</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Pedregosa</surname> <given-names>F.</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname> <given-names>A. C.</given-names>
</name>
<name>
<surname>Grisel</surname> <given-names>O.</given-names>
</name>
<etal/>
</person-group>. (<year>2013</year>). <article-title>API design for machine learning software: experiences from the scikit-learn project</article-title>. Available online at: <uri xlink:href="https://arxiv.org/abs/1309.0238v1">https://arxiv.org/abs/1309.0238v1</uri> (Accessed <access-date>June 24, 2024</access-date>).</citation></ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>XGBoost: A scalable tree boosting system</article-title>,&#x201d; in <source>
<italic>Proceedings of the ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</italic> 13-17-August-2016</source>, <fpage>785</fpage>&#x2013;<lpage>794</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/2939672.2939785</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Clevert</surname> <given-names>D. A.</given-names>
</name>
<name>
<surname>Unterthiner</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Hochreiter</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Fast and accurate deep network learning by exponential linear units (ELUs)</article-title>,&#x201d; in <conf-name>4th International Conference on Learning Representations, ICLR 2016 - Conference Track Proceedings</conf-name>. Available at: <uri xlink:href="https://arxiv.org/abs/1511.07289v5">https://arxiv.org/abs/1511.07289v5</uri>.</citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Crossa</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Martini</surname> <given-names>J. W. R.</given-names>
</name>
<name>
<surname>Gianola</surname> <given-names>D.</given-names>
</name>
<name>
<surname>P&#xe9;rez-Rodr&#xed;guez</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Jarquin</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Juliana</surname> <given-names>P.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Deep kernel and deep learning for genome-based prediction of single traits in multienvironment breeding trials</article-title>. <source>Front. Genet.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/FGENE.2019.01168/BIBTEX</pub-id>, PMID: <pub-id pub-id-type="pmid">31921277</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cunha</surname> <given-names>R. L. F.</given-names>
</name>
<name>
<surname>Silva</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Netto</surname> <given-names>M. A. S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>A scalable machine learning system for pre-season agriculture yield forecast</article-title>,&#x201d; in <source>
<italic>Proceedings - IEEE 14th International Conference on eScience, e-Science</italic> 2018</source>, <fpage>423</fpage>&#x2013;<lpage>430</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ESCIENCE.2018.00131</pub-id>
</citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cutler</surname> <given-names>D. R.</given-names>
</name>
<name>
<surname>Edwards</surname> <given-names>T. C.</given-names>
</name>
<name>
<surname>Beard</surname> <given-names>K. H.</given-names>
</name>
<name>
<surname>Cutler</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hess</surname> <given-names>K. T.</given-names>
</name>
<name>
<surname>Gibson</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2007</year>). <article-title>RANDOM FORESTS FOR CLASSIFICATION IN ECOLOGY</article-title>. <source>Ecology</source> <volume>88</volume>, <fpage>2783</fpage>&#x2013;<lpage>2792</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1890/07-0539.1</pub-id>, PMID: <pub-id pub-id-type="pmid">18051647</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danilevicz</surname> <given-names>M. F.</given-names>
</name>
<name>
<surname>Gill</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Anderson</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Batley</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Bennamoun</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Bayer</surname> <given-names>P. E.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Plant genotype to phenotype prediction using machine learning</article-title>. <source>Front. Genet.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/FGENE.2022.822173/BIBTEX</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Darby</surname> <given-names>H. M.</given-names>
</name>
<name>
<surname>Lauer</surname> <given-names>J. G.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Planting date and hybrid influence on corn forage yield and quality</article-title>. <source>Agron. J.</source> <volume>94</volume>, <page-range>281&#x2013;289</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/agronj2002.0281</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dhaliwal</surname> <given-names>D. S.</given-names>
</name>
<name>
<surname>Williams</surname> <given-names>M. M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Sweet corn yield prediction using machine learning models and field-level data</article-title>. <source>Precis. Agric.</source> <volume>25</volume>, <fpage>51</fpage>&#x2013;<lpage>64</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/S11119-023-10057-1/FIGURES/3</pub-id>
</citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dobermann</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ping</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Adamchuk</surname> <given-names>V. I.</given-names>
</name>
<name>
<surname>Simbahan</surname> <given-names>G. C.</given-names>
</name>
<name>
<surname>Ferguson</surname> <given-names>R. B.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Classification of crop yield variability in irrigated production fields</article-title>. <source>Agron. J.</source> <volume>95</volume>, <fpage>1105</fpage>&#x2013;<lpage>1120</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/AGRONJ2003.1105</pub-id>
</citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dobor</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Barcza</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Hl&#xe1;sny</surname> <given-names>T.</given-names>
</name>
<name>
<surname>&#xc1;rend&#xe1;s</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Spitk&#xf3;</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Fodor</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Crop planting date matters: Estimation methods and effect on future yields</article-title>. <source>Agric. For. Meteorol.</source> <volume>223</volume>, <fpage>103</fpage>&#x2013;<lpage>115</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.AGRFORMET.2016.03.023</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elavarasan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Durairaj Vincent</surname> <given-names>P. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Crop yield prediction using deep reinforcement learning model for sustainable agrarian applications</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>86886</fpage>&#x2013;<lpage>86901</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2020.2992480</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Everingham</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Sexton</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Skocaj</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Inman-Bamber</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Accurate prediction of sugarcane yield using a random forest algorithm</article-title>. <source>Agron. Sustain. Dev.</source> <volume>36</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/S13593-016-0364-Z/FIGURES/3</pub-id>
</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fernandes</surname> <given-names>I. K.</given-names>
</name>
<name>
<surname>Vieira</surname> <given-names>C. C.</given-names>
</name>
<name>
<surname>Dias</surname> <given-names>K. O. G.</given-names>
</name>
<name>
<surname>Fernandes</surname> <given-names>S. B.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Using machine learning to integrate genetic and environmental data to model genotype-by-environment interactions</article-title>. <source>bioRxiv</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2024.02.08.579534</pub-id>. 2024.02.08.579534.</citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Filippi</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>E. J.</given-names>
</name>
<name>
<surname>Wimalathunge</surname> <given-names>N. S.</given-names>
</name>
<name>
<surname>Somarathna</surname> <given-names>P. D. S. N.</given-names>
</name>
<name>
<surname>Pozza</surname> <given-names>L. E.</given-names>
</name>
<name>
<surname>Ugbaje</surname> <given-names>S. U.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>An approach to forecast grain crop yield using multi-layered, multi-farm data sets and machine learning</article-title>. <source>Precis. Agric.</source> <volume>20</volume>, <fpage>1015</fpage>&#x2013;<lpage>1029</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/S11119-018-09628-4/FIGURES/5</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gambin</surname> <given-names>B. L.</given-names>
</name>
<name>
<surname>Coyos</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Di Mauro</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Borr&#xe1;s</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Garibaldi</surname> <given-names>L. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Exploring genotype, management, and environmental variables influencing grain yield of late-sown maize in central Argentina</article-title>. <source>Agric. Syst.</source> <volume>146</volume>, <fpage>11</fpage>&#x2013;<lpage>19</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.AGSY.2016.03.011</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<collab>Genomes to Fields</collab>
</person-group> (<year>2023</year>). <article-title>Genomes to fields 2022 dataset</article-title>. <source>CyVerse. Data Commons</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.25739/3d3g-pe51</pub-id>
</citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grinberg</surname> <given-names>N. F.</given-names>
</name>
<name>
<surname>Orhobor</surname> <given-names>O. I.</given-names>
</name>
<name>
<surname>King</surname> <given-names>R. D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An evaluation of machine-learning for predicting phenotype: studies in yeast, rice, and wheat</article-title>. <source>Mach. Learn.</source> <volume>109</volume>, <fpage>251</fpage>&#x2013;<lpage>277</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/S10994-019-05848-5/FIGURES/9</pub-id>, PMID: <pub-id pub-id-type="pmid">32174648</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hankinson</surname> <given-names>M. W.</given-names>
</name>
<name>
<surname>Lindsey</surname> <given-names>L. E.</given-names>
</name>
<name>
<surname>Culman</surname> <given-names>S. W.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Effect of planting date and starter fertilizer on soybean grain yield</article-title>. <source>Crop. Forage. Turfgrass Manage.</source> <volume>1</volume>, <page-range>1&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/cftm2015.0178</pub-id>
</citation></ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Haque</surname> <given-names>F. F.</given-names>
</name>
<name>
<surname>Abdelgawad</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Yanambaka</surname> <given-names>V. P.</given-names>
</name>
<name>
<surname>Yelamarthi</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Crop yield prediction using deep neural network</article-title>,&#x201d; in <source>IEEE World Forum on Internet of Things, WF-IoT 2020 - Symposium Proceedings</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/WF-IOT48130.2020.9221298</pub-id>, PMID: <pub-id pub-id-type="pmid">31191564</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heffner</surname> <given-names>E. L.</given-names>
</name>
<name>
<surname>Sorrells</surname> <given-names>M. E.</given-names>
</name>
<name>
<surname>Jannink</surname> <given-names>J. L.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Genomic Selection for crop improvement</article-title>. <source>Crop Sci.</source> <volume>49</volume>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2135/CROPSCI2008.08.0512</pub-id>
</citation></ref>
<ref id="B29">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>James</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Witten</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Hastie</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Tibshirani</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>An introduction to statistical learning</article-title>. <edition>2nd ed.</edition> (<publisher-loc>Newyork</publisher-loc>: <publisher-name>Springer</publisher-name>). Available online at: <uri xlink:href="http://www.springer.com/series/417">http://www.springer.com/series/417</uri>.</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaiser</surname> <given-names>D. E.</given-names>
</name>
<name>
<surname>Coulter</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Vetsch</surname> <given-names>J. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Corn hybrid response to in-furrow starter fertilizer as affected by planting date</article-title>. <source>Agron. J.</source> <volume>108</volume>, <page-range>2493&#x2013;2501</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/agronj2016.02.0124</pub-id>
</citation></ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khaki</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Crop yield prediction using deep neural networks</article-title>. <source>Front. Plant Sci.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/FPLS.2019.00621/BIBTEX</pub-id>
</citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khaki</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Archontoulis</surname> <given-names>S. V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A CNN-RNN framework for crop yield prediction</article-title>. <source>Front. Plant Sci.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/FPLS.2019.01750/BIBTEX</pub-id>, PMID: <pub-id pub-id-type="pmid">32038699</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kolipaka</surname> <given-names>V. R. R.</given-names>
</name>
<name>
<surname>Namburu</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An automatic crop yield prediction framework designed with two-stage classifiers: a meta-heuristic approach</article-title>. <source>Multimed. Tools Appl.</source> <volume>83</volume>, <fpage>28969</fpage>&#x2013;<lpage>28992</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/S11042-023-16612-2/METRICS</pub-id>
</citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kouadio</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Deo</surname> <given-names>R. C.</given-names>
</name>
<name>
<surname>Byrareddy</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Adamowski</surname> <given-names>J. F.</given-names>
</name>
<name>
<surname>Mushtaq</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Phuong Nguyen</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Artificial intelligence approach for the prediction of Robusta coffee yield using soil fertility properties</article-title>. <source>Comput. Electron. Agric.</source> <volume>155</volume>, <fpage>324</fpage>&#x2013;<lpage>338</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.COMPAG.2018.10.014</pub-id>
</citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lauer</surname> <given-names>J. G.</given-names>
</name>
<name>
<surname>Carter</surname> <given-names>P. R.</given-names>
</name>
<name>
<surname>Wood</surname> <given-names>T. M.</given-names>
</name>
<name>
<surname>Diezel</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Wiersma</surname> <given-names>D. W.</given-names>
</name>
<name>
<surname>Rand</surname> <given-names>R. E.</given-names>
</name>
<etal/>
</person-group>. (<year>1999</year>). <article-title>Corn hybrid response to planting date in the northern corn belt</article-title>. <source>Agron. J.</source> <volume>91</volume>, <page-range>834&#x2013;839</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/agronj1999.915834x</pub-id>
</citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lima</surname> <given-names>D. C.</given-names>
</name>
<name>
<surname>Washburn</surname> <given-names>J. D.</given-names>
</name>
<name>
<surname>Varela</surname> <given-names>J. I.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Gage</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Romay</surname> <given-names>M. C.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Genomes to fields 2022 maize genotype by environment prediction competition</article-title>. <source>BMC Res. Notes</source> <volume>16</volume>, <fpage>1</fpage>&#x2013;<lpage>3</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/S13104-023-06421-Z/TABLES/1</pub-id>
</citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lopez-Cruz</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Aguate</surname> <given-names>F. M.</given-names>
</name>
<name>
<surname>Washburn</surname> <given-names>J. D.</given-names>
</name>
<name>
<surname>De Leon</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Kaeppler</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Lima</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Leveraging data from the Genomes-to-Fields Initiative to investigate genotype-by-environment interactions in maize in North America</article-title>. <source>Nat. Commun.</source> <volume>14</volume>, <fpage>1</fpage>&#x2013;<lpage>14</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-023-42687-4</pub-id>, PMID: <pub-id pub-id-type="pmid">37903778</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>A deep convolutional neural network approach for predicting phenotypes from genotypes</article-title>. <source>Planta</source> <volume>248</volume>, <fpage>1307</fpage>&#x2013;<lpage>1318</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/S00425-018-2976-9/METRICS</pub-id>
</citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maestrini</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Basso</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Predicting spatial patterns of within-field crop yield variability</article-title>. <source>Field Crops Res.</source> <volume>219</volume>, <fpage>106</fpage>&#x2013;<lpage>112</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.FCR.2018.01.028</pub-id>
</citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maimaitijiang</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sagan</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Sidike</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Hartling</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Esposito</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Fritschi</surname> <given-names>F. B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Soybean yield prediction from UAV using multimodal data fusion and deep learning</article-title>. <source>Remote Sens. Environ.</source> <volume>237</volume>, <elocation-id>111599</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.RSE.2019.111599</pub-id>
</citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Masud Rana</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Belal Hossain</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kumar Roy</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Shultana</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Rokebul Hasan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Aminun Naher</surname> <given-names>U.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Response of Yield and Agronomic Output of Bangabandhu dhan100 under Varying Sowing Window in Cold Prone Rangpur Region</article-title>. <source>Indian J. Agric. Res.</source> <volume>58</volume>, <fpage>259</fpage>&#x2013;<lpage>265</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.18805/IJARe.AF-796</pub-id>
</citation></ref>
<ref id="B42">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Medsker</surname> <given-names>L. R.</given-names>
</name>
<name>
<surname>Jain</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Recurrent neural networks</article-title>. <source>Design and Applications</source>.  <volume>5</volume> (<issue>64-67</issue>), <fpage>2</fpage>.</citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mia</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Tanabe</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Habibi</surname> <given-names>L. N.</given-names>
</name>
<name>
<surname>Hashimoto</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Homma</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Maki</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Multimodal deep learning for rice yield prediction using UAV-based multispectral imagery and weather data</article-title>. <source>Remote Sens. (Basel).</source> <volume>15</volume>, <elocation-id>2511</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/RS15102511/S1</pub-id>
</citation></ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mir</surname> <given-names>R. R.</given-names>
</name>
<name>
<surname>Reynolds</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Pinto</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Bhat</surname> <given-names>M. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>High-throughput phenotyping for crop improvement in the genomics era</article-title>. <source>Plant Sci.</source> <volume>282</volume>, <fpage>60</fpage>&#x2013;<lpage>72</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.PLANTSCI.2019.01.007</pub-id>, PMID: <pub-id pub-id-type="pmid">31003612</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Monroe</surname> <given-names>J. G.</given-names>
</name>
<name>
<surname>Arciniegas</surname> <given-names>J. P.</given-names>
</name>
<name>
<surname>Moreno</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>S&#xe1;nchez</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Sierra</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Valdes</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>The lowest hanging fruit: Beneficial gene knockouts in past, present, and future crop evolution</article-title>. <source>Curr. Plant Biol.</source> <volume>24</volume>, <elocation-id>100185</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.CPB.2020.100185</pub-id>
</citation></ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Montesinos-L&#xf3;pez</surname> <given-names>O. A.</given-names>
</name>
<name>
<surname>Montesinos-L&#xf3;pez</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Crossa</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Gianola</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Hern&#xe1;ndez-Su&#xe1;rez</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>Mart&#xed;n-Vallejo</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>b). <article-title>Multi-trait, multi-environment deep learning modeling for genomic-enabled prediction of plant traits</article-title>. <source>G3 Genes|Genomes|Genet.</source> <volume>8</volume>, <fpage>3829</fpage>&#x2013;<lpage>3840</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1534/G3.118.200728</pub-id>, PMID: <pub-id pub-id-type="pmid">30291108</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Montesinos-L&#xf3;pez</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Montesinos-L&#xf3;pez</surname> <given-names>O. A.</given-names>
</name>
<name>
<surname>Gianola</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Crossa</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hern&#xe1;ndez-Su&#xe1;rez</surname> <given-names>C. M.</given-names>
</name>
</person-group> (<year>2018</year>a). <article-title>Multi-environment genomic prediction of plant traits using deep learners with dense architecture</article-title>. <source>G3 Genes|Genomes|Genet.</source> <volume>8</volume>, <fpage>3813</fpage>&#x2013;<lpage>3828</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1534/G3.118.200740</pub-id>, PMID: <pub-id pub-id-type="pmid">30291107</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moseley</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Reis</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Gentimis</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Campos</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Copes</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Netterville</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Soybean planting dates and maturity groups: Maximizing yield potential and decreasing risk in Louisiana</article-title>. <source>Agron. J</source>. <volume>116</volume> (<issue>5</issue>), <page-range>2174&#x2013;2185</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/AGJ2.21626</pub-id>
</citation></ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mrubata</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Nciizah</surname> <given-names>A. D.</given-names>
</name>
<name>
<surname>Muchaonyerwa</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Planting date and tillage effects on yield and nutrient uptake of two sorghum cultivars grown in sub-humid and semi-arid regions in South Africa</article-title>. <source>Front. Agron.</source> <volume>6</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/FAGRO.2024.1388823/BIBTEX</pub-id>
</citation></ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nevavuori</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Narra</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Lipping</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Crop yield prediction with deep convolutional neural networks</article-title>. <source>Comput. Electron. Agric.</source> <volume>163</volume>, <elocation-id>104859</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.COMPAG.2019.104859</pub-id>
</citation></ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nuccio</surname> <given-names>M. L.</given-names>
</name>
<name>
<surname>Paul</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Bate</surname> <given-names>N. J.</given-names>
</name>
<name>
<surname>Cohn</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cutler</surname> <given-names>S. R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Where are the drought tolerant crops? An assessment of more than two decades of plant biotechnology effort in crop improvement</article-title>. <source>Plant Sci.</source> <volume>273</volume>, <fpage>110</fpage>&#x2013;<lpage>119</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.PLANTSCI.2018.01.020</pub-id>, PMID: <pub-id pub-id-type="pmid">29907303</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oakey</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Cullis</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Thompson</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Comadran</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Halpin</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Waugh</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Genomic selection in multi-environment crop trials</article-title>. <source>G3.: Genes. Genomes. Genet.</source> <volume>6</volume>, <fpage>1313</fpage>&#x2013;<lpage>1326</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1534/G3.116.027524/-/DC1</pub-id>
</citation></ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oikonomidis</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Catal</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Kassahun</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Hybrid deep learning-based models for crop yield prediction</article-title>. <source>Appl. Artif. Intell.</source> <volume>36</volume> (<issue>1</issue>), <elocation-id>2031822</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/08839514.2022.2031823/FORMAT/EPUB</pub-id>
</citation></ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paudel</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Boogaard</surname> <given-names>H.</given-names>
</name>
<name>
<surname>de Wit</surname> <given-names>A.</given-names>
</name>
<name>
<surname>van der Velde</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Claverie</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Nisini</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Machine learning for regional crop yield forecasting in Europe</article-title>. <source>Field Crops Res.</source> <volume>276</volume>, <elocation-id>108377</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.FCR.2021.108377</pub-id>
</citation></ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sajid</surname> <given-names>S. S.</given-names>
</name>
<name>
<surname>Shahhosseini</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Huber</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Archontoulis</surname> <given-names>S. V.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>County-scale crop yield prediction by integrating crop simulation with machine learning models</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/FPLS.2022.1000224/BIBTEX</pub-id>, PMID: <pub-id pub-id-type="pmid">36518505</pub-id></citation></ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sarzaeim</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Mu&#xf1;oz-Arriola</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A method to estimate climate drivers of maize yield predictability leveraging genetic-by-environment interactions in the US and Canada</article-title>. <source>Agronomy</source> <volume>14</volume>, <elocation-id>733</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/AGRONOMY14040733</pub-id>
</citation></ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scheben</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wolter</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Batley</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Puchta</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Edwards</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Towards CRISPR/Cas crops &#x2013; bringing together genomics and genome editing</article-title>. <source>New Phytol.</source> <volume>216</volume>, <fpage>682</fpage>&#x2013;<lpage>698</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/NPH.14702</pub-id>, PMID: <pub-id pub-id-type="pmid">28762506</pub-id></citation></ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shahhosseini</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Archontoulis</surname> <given-names>S. V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Forecasting corn yield with machine learning ensembles</article-title>. <source>Front. Plant Sci.</source> <volume>11</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2020.01120</pub-id>, PMID: <pub-id pub-id-type="pmid">32849688</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shahhosseini</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Huber</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Archontoulis</surname> <given-names>S. V.</given-names>
</name>
</person-group> (<year>2021</year>a). <article-title>Coupling machine learning and crop modeling improves crop yield prediction in the US Corn Belt</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-020-80820-1</pub-id>, PMID: <pub-id pub-id-type="pmid">33452349</pub-id></citation></ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shahhosseini</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Khaki</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Archontoulis</surname> <given-names>S. V.</given-names>
</name>
</person-group> (<year>2021</year>b). <article-title>Corn yield prediction with ensemble CNN-DNN</article-title>. <source>Front. Plant Sci.</source> <volume>12</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/FPLS.2021.709008/FULL</pub-id>, PMID: <pub-id pub-id-type="pmid">34408763</pub-id></citation></ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shook</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Gangopadhyay</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Ganapathysubramanian</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Sarkar</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>A. K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Crop yield prediction integrating genotype and weather variables using deep learning</article-title>. <source>PloS One</source> <volume>16</volume>, <elocation-id>e0252402</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/JOURNAL.PONE.0252402</pub-id>, PMID: <pub-id pub-id-type="pmid">34138872</pub-id></citation></ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Swanson</surname> <given-names>S. P.</given-names>
</name>
<name>
<surname>Wilhelm</surname> <given-names>W. W.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Planting date and residue rate effects on growth, partitioning, and yield of corn</article-title>. <source>Agron. J.</source> <volume>88</volume>, <page-range>205&#x2013;210</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/agronj1996.00021962008800020014x</pub-id>
</citation></ref>
<ref id="B63">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>The Genomes To Fields Initiative</collab>
</person-group> (<year>2023</year>). Available online at: <uri xlink:href="https://www.genomes2fields.org/">https://www.genomes2fields.org/</uri> (Accessed <access-date>November 15, 2023</access-date>).</citation></ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thorburn</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Dietzel</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Cammarano</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Ciampitti</surname> <given-names>I. A.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>N. V.</given-names>
</name>
<name>
<surname>Assefa</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Maize yield and planting date relationship: A synthesis-analysis for US high-yielding contest-winner and field research data</article-title>. <source> Front. Plant Sci.</source> <volume>8</volume>, <fpage>2106</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2017.02106</pub-id>, PMID: <pub-id pub-id-type="pmid">29312377</pub-id></citation></ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tibshiranit</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Regression shrinkage and selection via the lasso</article-title>. <source>J. R. Stat. Soc.: Ser. B. (Methodological).</source> <volume>58</volume>, <fpage>267</fpage>&#x2013;<lpage>288</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/J.2517-6161.1996.TB02080.X</pub-id>
</citation></ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsimba</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Edmeades</surname> <given-names>G. O.</given-names>
</name>
<name>
<surname>Millner</surname> <given-names>J. P.</given-names>
</name>
<name>
<surname>Kemp</surname> <given-names>P. D.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>The effect of planting date on maize grain yields and yield components</article-title>. <source>Field Crops Res.</source> <volume>150</volume>, <fpage>135</fpage>&#x2013;<lpage>144</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.FCR.2013.05.028</pub-id>
</citation></ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Roekel</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Coulter</surname> <given-names>J. A.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Agronomic responses of corn to planting date and plant density</article-title>. <source>Agron. J.</source> <volume>103</volume>, <fpage>1414</fpage>&#x2013;<lpage>1422</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2134/agronj2011.0071</pub-id>
</citation></ref>
<ref id="B68">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>A. X.</given-names>
</name>
<name>
<surname>Tran</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Desai</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Lobell</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Ermon</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Deep transfer learning for crop yield prediction with remote sensing data</article-title>,&#x201d; in <source>Proceedings of the 1st ACM SIGCAS Conference on Computing and Sustainable Societies, COMPASS</source>, vol. <volume>18</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3209811.3212707</pub-id>
</citation></ref>
<ref id="B69">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Washburn</surname> <given-names>J. D.</given-names>
</name>
<name>
<surname>Varela</surname> <given-names>J. I.</given-names>
</name>
<name>
<surname>Xavier</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Ertl</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Gage</surname> <given-names>J. L.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Global genotype by environment prediction competition reveals that diverse modeling strategies can deliver satisfactory maize yield estimates</article-title>. <source>bioRxiv</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2024.09.13.612969</pub-id>. 2024.09.13.612969., PMID: <pub-id pub-id-type="pmid">39345633</pub-id></citation></ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Williams</surname> <given-names>M. M.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Planting date influences critical period of weed control in sweet corn</article-title>. <source>Weed. Sci.</source> <volume>54</volume>, <fpage>928</fpage>&#x2013;<lpage>933</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1614/ws-06-005r.1</pub-id>
</citation></ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Weng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>He</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Time-series&#xa0;field&#xa0;phenotyping of soybean growth analysis by combining multimodal&#xa0;deep&#xa0;learning and dynamic modeling</article-title>. <source>Plant Phenomics.</source> <volume>6</volume>, <elocation-id>0158</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/PLANTPHENOMICS.0158/SUPPL_FILE/PLANTPHENOMICS.0158.F1.ZIP</pub-id>, PMID: <pub-id pub-id-type="pmid">38524738</pub-id></citation></ref>
</ref-list>
</back>
</article>