<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2023.1234555</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Ensemble machine learning-based recommendation system for effective prediction of suitable agricultural crop cultivation</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Hasan</surname>
<given-names>Mahmudul</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/727358"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Marjan</surname>
<given-names>Md Abu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Uddin</surname>
<given-names>Md Palash</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2008775"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Afjal</surname>
<given-names>Masud Ibn</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kardy</surname>
<given-names>Seifedine</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2086118"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Shaoqi</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Nam</surname>
<given-names>Yunyoung</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1592882"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Computer Science and Engineering, Hajee Mohammad Danesh Science and Technology University</institution>, <addr-line>Dinajpur</addr-line>, <country>Bangladesh</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Information Technology, Deakin University</institution>, <addr-line>Geelong, VIC</addr-line>, <country>Australia</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Applied Data Science, Noroff University College</institution>, <addr-line>Kristiansand</addr-line>, <country>Norway</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Artificial Intelligence Research Center (AIRC), Ajman University</institution>, <addr-line>Ajman</addr-line>, <country>United Arab Emirates</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Electrical and Computer Engineering, Lebanese American University</institution>, <addr-line>Byblos</addr-line>, <country>Lebanon</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of ICT Convergence, Soonchunhyang University</institution>, <addr-line>Asan</addr-line>, <country>Republic of Korea</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Muhammad Fazal Ijaz, Sejong University, Republic of Korea</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Sambit Bakshi, National Institute of Technology Rourkela, India; Mohammed Chachan Younis, University of Mosul, Iraq; Yassine Maleh, Universit&#xe9; Sultan Moulay Slimane, Morocco</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Seifedine Kardy, <email xlink:href="mailto:seifedine.kadry@noroff.no">seifedine.kadry@noroff.no</email>; Yunyoung Nam, <email xlink:href="mailto:ynam@sch.ac.kr">ynam@sch.ac.kr</email>; Md Palash Uddin, <email xlink:href="mailto:palash_cse@hstu.ac.bd">palash_cse@hstu.ac.bd</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>10</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1234555</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>06</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>07</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Hasan, Marjan, Uddin, Afjal, Kardy, Ma and Nam</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Hasan, Marjan, Uddin, Afjal, Kardy, Ma and Nam</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Agriculture is the most critical sector for food supply on the earth, and it is also responsible for supplying raw materials for other industrial productions. Currently, the growth in agricultural production is not sufficient to keep up with the growing population, which may result in a food shortfall for the world&#x2019;s inhabitants. As a result, increasing food production is crucial for developing nations with limited land and resources. It is essential to select a suitable crop for a specific region to increase its production rate. Effective crop production forecasting in that area based on historical data, including environmental and cultivation areas, and crop production amount, is required. However, the data for such forecasting are not publicly available. As such, in this paper, we take a case study of a developing country, Bangladesh, whose economy relies on agriculture. We first gather and preprocess the data from the relevant research institutions of Bangladesh and then propose an ensemble machine learning approach, called K-nearest Neighbor Random Forest Ridge Regression (KRR), to effectively predict the production of the major crops (three different kinds of rice, potato, and wheat). KRR is designed after investigating five existing traditional machine learning (Support Vector Regression, Na&#xef;ve Bayes, and Ridge Regression) and ensemble learning (Random Forest and CatBoost) algorithms. We consider four classical evaluation metrics, i.e., mean absolute error, mean square error (MSE), root MSE, and <italic>R</italic>
<sup>2</sup>, to evaluate the performance of the proposed KRR over the other machine learning models. It shows 0.009 MSE, 99% <italic>R</italic>
<sup>2</sup> for Aus; 0.92 MSE, 90% <italic>R</italic>
<sup>2</sup> for Aman; 0.246 MSE, 99% <italic>R</italic>
<sup>2</sup> for Boro; 0.062 MSE, 99% <italic>R</italic>
<sup>2</sup> for wheat; and 0.016 MSE, 99% <italic>R</italic>
<sup>2</sup> for potato production prediction. The Diebold&#x2013;Mariano test is conducted to check the robustness of the proposed ensemble model, KRR. In most cases, it shows 1% and 5% significance compared to the benchmark ML models. Lastly, we design a recommender system that suggests suitable crops for a specific land area for cultivation in the next season. We believe that the proposed paradigm will help the farmers and personnel in the agricultural sector leverage proper crop cultivation and production.</p>
</abstract>
<kwd-group>
<kwd>crop production</kwd>
<kwd>crop prediction</kwd>
<kwd>agricultural data processing</kwd>
<kwd>machine learning</kwd>
<kwd>ensemble learning</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="10"/>
<equation-count count="4"/>
<ref-count count="65"/>
<page-count count="18"/>
<word-count count="9499"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Plant Bioinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<title>Introduction</title>
<p>A constructive agricultural environment and fertile land make agriculture the leading economic sector for a developing country whose economy relies on agriculture. Agriculture is associated with producing essential food crops and industrial raw materials. One of the most critical aspects of the development cycle of a country is the capacity to produce food using the unfavorable environment and limited agricultural land (<xref ref-type="bibr" rid="B23">Goldstein et&#xa0;al., 2017</xref>). Experts believe that land fertility has reduced to a certain extent over time, affecting the crop production amount (<xref ref-type="bibr" rid="B61">Van Klompenburg et&#xa0;al., 2020</xref>). In this paper, we consider the case study of a developing country, Bangladesh, whose economy relies on agriculture. According to the Bangladesh Rural Advancement Committee (BRAC), the agricultural land in Bangladesh is shrinking by 1% annually, while the population is growing by 1.2% annually (<xref ref-type="bibr" rid="B16">Das et al., 2022</xref>). In addition, the farmers do not get the actual price due to the lack of knowledge of the estimated crop production. This concern demotivates the farmers, which has a long-term negative impact on the agriculture sector. To alleviate this issue, proper planning of the best crop production in terms of correctly predicting crop production for the upcoming year can be provided to the farmers. The ability to accurately predict crop yields has become essential for farmers to make rational choices (<xref ref-type="bibr" rid="B28">Jansson et&#xa0;al., 2021</xref>). Various aspects, such as soil type, weather, and crop management practices, are taken into account to estimate the number of crops that may be grown in a particular area. Effective prediction helps to generate an estimation of crops that helps the government to take long-term and short-term policies to minimize food shortages and import&#x2013;export plans based on the agriculture sector (<xref ref-type="bibr" rid="B65">Zhang et&#xa0;al., 2019</xref>). It also significantly impacts the economy of an agricultural-based country like our study area. Machine learning (ML) offers the most effective tool to predict the dependent variables (i.e., crop production) using the independent variables (i.e., the factors that regulate crop production) (<xref ref-type="bibr" rid="B29">Jayalakshmi and Gomathi, 2020</xref>; <xref ref-type="bibr" rid="B1">Ahmed et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B35">Li et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B37">Monteiro et&#xa0;al., 2022</xref>). In this paper, we investigate the simple but effective ML approaches to propose an ensemble ML approach toward accurately predicting the agricultural crop production of Bangladesh.</p>
<p>Bangladesh is a country with six seasons, which enables producing different kinds of crops over the year (<xref ref-type="bibr" rid="B59">Uddin et&#xa0;al., 2019</xref>) while its main crops are rice, wheat, and potato. Rice is the staple crop, and it can be cultivated in three different seasons where the rice varieties are Aus, Aman, and Boro. Potato and wheat are the second and third most important crops, respectively. As such, we predict the production of these five major crops (Aus rice, Aman rice, Boro rice, potato, and wheat) for the upcoming season based on the environmental data (i.e., rainfall, humidity, minimum and maximum temperature, sunshine, wind speed, and cloud coverage of a specific zone), cultivation area, and previous production data. We use historical data from 1969 to 2021 of different districts of Bangladesh (<xref ref-type="bibr" rid="B9">Campbell et&#xa0;al., 2020</xref>) and collect these raw data from different respective government organizations. In particular, we gather the raw data from the yearbooks of the Bangladesh Meteorological Department (BMD), Bangladesh Agricultural Development Corporation (BADC), Bangladesh Rice Research Institute (BRRI), and Bangladesh Bureau of Statistics (BBS). After that, we investigate the classical ML algorithms, i.e., Support Vector Regression (SVR), Na&#xef;ve Bayes (NB), and Ridge Regression (RR), and ensemble ML algorithms, i.e., Random Forest (RF) and CatBoost (CB). Then, we propose an ensemble ML paradigm combining K-Nearest Neighbors (KNN), RF, and RR, termed K-nearest neighbors Random Forest Ridge regression (KRR), to effectively predict the production of the crops. Finally, we construct a recommender system that suggests suitable crops for a given land area for cultivation in the next season. The main contributions of this paper are summarized below.</p>
<list list-type="bullet">
<list-item>
<p>Development (collection, reformation, and data processing) of an ML trainable crop dataset containing environmental, cultivation area, and previous production data for predicting five major crops (Aus rice, Aman rice, Boro rice, potato, and wheat);</p>
</list-item>
<list-item>
<p>Investigation and rigorous study of setting up a baseline ML system with effective ML and ensemble ML models for predicting crop production more efficiently;</p>
</list-item>
<list-item>
<p>Design of a novel ensemble ML algorithm to accurately predict the production of the crops and Diebold&#x2013;Mariano (DM) testing of the designed ensemble ML model to illustrate its significance and superiority over the benchmark ML and ensemble ML algorithms; and</p>
</list-item>
<list-item>
<p>Designing a recommendation system for suggesting suitable crops for cultivating in a specific region in the next season among the contemporary crops.</p>
</list-item>
</list>
<p>The rest of this paper is structured as follows. In the <italic>Related work</italic> section, we discuss and compare the related works on crop production prediction. The <italic>Proposed paradigm</italic> section describes the overall idea and development of the proposed crop production prediction and recommendation paradigm. In the <italic>Methods and measurements</italic> section, we discuss the methods and materials for dataset generation, existing ML and ensemble ML methods, and our proposed ensemble ML approach. The experiments and results are explained and analyzed in the <italic>Experiment and result analysis</italic> section. The <italic>Crop recommender system</italic> section demonstrates the recommender system design for suggesting suitable crop cultivation in the upcoming season, while the <italic>Conclusion</italic> section summarizes and concludes the observations and findings. All the abbreviations used in this paper are listed in <xref ref-type="table" rid="T1">
<bold>Table 1</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>List of the abbreviations used in the paper.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Abbreviations</th>
<th valign="top" align="left">Full Form</th>
<th valign="top" align="left">Abbreviations</th>
<th valign="top" align="left">Full Form</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">KRR</td>
<td valign="top" align="left">K-nearest Neighbor Random Forest Ridge Regression</td>
<td valign="top" align="left">BRAC</td>
<td valign="top" align="left">Bangladesh Rural Advancement Committee</td>
</tr>
<tr>
<td valign="top" align="left">ML</td>
<td valign="top" align="left">Machine Learning</td>
<td valign="top" align="left">BMD</td>
<td valign="top" align="left">Bangladesh Meteorological Department</td>
</tr>
<tr>
<td valign="top" align="left">BADC</td>
<td valign="top" align="left">Bangladesh Agricultural Development Corporation</td>
<td valign="top" align="left">BRRI</td>
<td valign="top" align="left">Bangladesh Rice Research Institute</td>
</tr>
<tr>
<td valign="top" align="left">BBS</td>
<td valign="top" align="left">Bangladesh Bureau of Statistics</td>
<td valign="top" align="left">SVR</td>
<td valign="top" align="left">Support Vector Regression</td>
</tr>
<tr>
<td valign="top" align="left">NB</td>
<td valign="top" align="left">Naive Bayes</td>
<td valign="top" align="left">RR</td>
<td valign="top" align="left">Ridge Regression</td>
</tr>
<tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">Random Forest</td>
<td valign="top" align="left">CB</td>
<td valign="top" align="left">CatBoost</td>
</tr>
<tr>
<td valign="top" align="left">KNN</td>
<td valign="top" align="left">K-Nearest Neighbors</td>
<td valign="top" align="left">DM</td>
<td valign="top" align="left">Diebold&#x2013;Mariano</td>
</tr>
<tr>
<td valign="top" align="left">ARIMA</td>
<td valign="top" align="left">Auto-regressive Integrated Moving Average</td>
<td valign="top" align="left">SVM</td>
<td valign="top" align="left">Support&#x2003;Vector Machine</td>
</tr>
<tr>
<td valign="top" align="left">APC</td>
<td valign="top" align="left">Average Pearson Correlation</td>
<td valign="top" align="left">CV</td>
<td valign="top" align="left">Coefficient of Variance</td>
</tr>
<tr>
<td valign="top" align="left">NN</td>
<td valign="top" align="left">Neural Network</td>
<td valign="top" align="left">LR</td>
<td valign="top" align="left">Logistic Regression</td>
</tr>
<tr>
<td valign="top" align="left">MSE</td>
<td valign="top" align="left">Mean Square Error</td>
<td valign="top" align="left">RMSE</td>
<td valign="top" align="left">Root Mean Square Error</td>
</tr>
<tr>
<td valign="top" align="left">MAE</td>
<td valign="top" align="left">Mean Absolute Error</td>
<td valign="top" align="left">RR</td>
<td valign="top" align="left">Ridge Regression</td>
</tr>
<tr>
<td valign="top" align="left">DT</td>
<td valign="top" align="left">Decision Tree</td>
<td valign="top" align="left">DL</td>
<td valign="top" align="left">Deep Learning</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2">
<title>Related work</title>
<p>Various applications of ML models in agriculture have been listed, such as crop yield prediction, weather forecasting, smart irrigation system, crop disease prediction, and deciding minimum support price (<xref ref-type="bibr" rid="B2">Al-Gaadi et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B39">Nandy and Singh, 2020</xref>; <xref ref-type="bibr" rid="B52">Sharma et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B14">Cravero and Sepulveda, 2021</xref>). Moreover, in order to achieve accurate predictions, researchers used the supervised ML algorithms for crop production prediction in (<xref ref-type="bibr" rid="B32">Kaur, 2016</xref>; <xref ref-type="bibr" rid="B53">Shehadeh et&#xa0;al., 2021</xref>). The decision tree (DT) classifier has been used to create predictions of yield and cropland temperature in (<xref ref-type="bibr" rid="B34">Lee and Moon, 2014</xref>) (<xref ref-type="bibr" rid="B3">Bagis et&#xa0;al., 2012</xref>). KNN and ID3 (a variant of DT) were applied to analyze the crop production of the previous year (<xref ref-type="bibr" rid="B12">Charbuty and Abdulazeez, 2021</xref>). Many researchers are using statistical models like Auto-regressive Integrated Moving Average (ARIMA) and ML model Support Vector Machine (SVM) for predicting crop production (<xref ref-type="bibr" rid="B57">Sujjaviriyasup and Pitiruek, 2013</xref>). On the other hand, time series analysis has been applied in order to predict the production and the price of crops and vegetables. The aim was to identify a time series function, which might identify patterns and seasonality in specific vegetables, as well as explore supply and demand variables (<xref ref-type="bibr" rid="B3">Bagis et&#xa0;al., 2012</xref>) (<xref ref-type="bibr" rid="B30">Jha and Sinha, 2013</xref>) (<xref ref-type="bibr" rid="B63">Young, 2019</xref>). In addition, many researchers proposed a methodology that uses Average Pearson Correlation (APC) and Coefficient of Variance (CV) to determine indications that reveal crop price fluctuation (<xref ref-type="bibr" rid="B42">Pereira et&#xa0;al., 2021</xref>). All these methods require the dataset to be extremely clearly described, which is difficult to generate in the context of Bangladesh.</p>
<p>Recently, satellite data have been utilized to predict the temperature in crop-growing areas (<xref ref-type="bibr" rid="B43">Prasad et&#xa0;al., 2021</xref>) (<xref ref-type="bibr" rid="B15">Danilevicz et&#xa0;al., 2021</xref>) (<xref ref-type="bibr" rid="B31">Jung et&#xa0;al., 2021</xref>). Because this method requires access to real-time satellite data, it would be inaccessible to most people. The precision of this method was additionally found to be insufficient. Some researchers also used the Neural Network (NN) approach to predict crop production, which might perform better than traditional ML methods (<xref ref-type="bibr" rid="B36">Minghua et&#xa0;al., 2012</xref>). However, NN is most common when working with multidimensional data. When the types of datasets are defined, the network model becomes more difficult to design, and more training time is required as the convergence time increases. It is also prone to slipping into the local minimal state.</p>
<p>Researchers have devised a way to predict crop yields at multiple spatial levels based on ML crop yield forecasts for regions. They developed a general ML workflow to show how proper regional agricultural yield forecasting can be in Europe. They predicted crop yields for 35 case studies, comprising nine nations that are major producers of six commodities (soft wheat, spring barley, sunflower, grain maize, sugar beets, and potatoes), to evaluate the validity and usefulness of regional predictions (<xref ref-type="bibr" rid="B41">Paudel et&#xa0;al., 2022</xref>). For the prediction of Irish potato and maize, authors collected data from multiple areas and analyzed it using RF, Polynomial Regression, and the SVR. The only variables employed as forecasters were rainfall and temperature. RMSE for RF was 510.8 and 129.9 for potato and maize, respectively, while <italic>R</italic>
<sup>2</sup> was 0.875 and 0.817 for the same crop datasets, indicating that RF is the best model (<xref ref-type="bibr" rid="B33">Kuradusenge et&#xa0;al., 2023</xref>). Based on the previous 12 years&#x2019; data, researchers proposed an ML-based crop yield prediction in North China Plan. To find the best model, they investigate several ML algorithms on winter wheat and dry matter prediction (<xref ref-type="bibr" rid="B62">Wang et&#xa0;al., 2023</xref>).</p>
<p>In comparison to the existing works in the literature, we, in our study, (i) generate a learnable environmental dataset containing eight features to predict crop production; (ii) propose an ensemble ML algorithm using KNN, RF, and RR, called KRR; (iii) demonstrate that KRR produces better results than the other classical ML algorithms, such as SVR, NB, RR, RF, and CB; and (iv) design a recommender system that suggests suitable crops to grow in the next session. To this end, we deliver the parametric differences between our proposed paradigm and the studied interconnected crop production prediction works in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison among the related works on crop production prediction.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Ref</th>
<th valign="top" align="left">Year</th>
<th valign="top" align="left">Dataset</th>
<th valign="top" align="left">Technique</th>
<th valign="top" align="left">Error/Score</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">(<xref ref-type="bibr" rid="B14">Cravero and Sepulveda, 2021</xref>)</td>
<td valign="top" align="left">2021</td>
<td valign="top" align="left">Big data</td>
<td valign="top" align="left">Classical ML and ensemble ML</td>
<td valign="top" align="left">Comparison charts</td>
</tr>
<tr>
<td valign="top" align="center">(<xref ref-type="bibr" rid="B39">Nandy and Singh, 2020</xref>)</td>
<td valign="top" align="left">2020</td>
<td valign="top" align="left">Collected data using multistage random sampling technique from 45 rural areas in West Bengal of India</td>
<td valign="top" align="left">RF and Logistic Regression (LR)</td>
<td valign="top" align="left">RF = 75.21% accuracy and 85.0% AUC and LR = 72.34% accuracy and 78.0% AUC</td>
</tr>
<tr>
<td valign="top" align="center">(<xref ref-type="bibr" rid="B52">Sharma et&#xa0;al., 2020</xref>)</td>
<td valign="top" align="left">2020</td>
<td valign="top" align="left">Collecting data from different sources</td>
<td valign="top" align="left">RF, DT, Bayesian network, SVM, NN, and GA</td>
<td valign="top" align="left">Comparison among the models</td>
</tr>
<tr>
<td valign="top" align="center">(<xref ref-type="bibr" rid="B36">Minghua et&#xa0;al., 2012</xref>)</td>
<td valign="top" align="left">2012</td>
<td valign="top" align="left">Historical agricultural product price data in China</td>
<td valign="top" align="left">NN</td>
<td valign="top" align="left">Prediction error 6.5% and 8.1% for different years</td>
</tr>
<tr>
<td valign="top" align="center">(<xref ref-type="bibr" rid="B53">Shehadeh et&#xa0;al., 2021</xref>)</td>
<td valign="top" align="left">2020</td>
<td valign="top" align="left">Bureau of Economic Analysis, U.S. Census Bureau</td>
<td valign="top" align="left">DT, LightGBM, and XGBoost</td>
<td valign="top" align="left">DT shows 93% accuracy, LightGBM 87%, and XGBoost 85%</td>
</tr>
<tr>
<td valign="top" align="center">(<xref ref-type="bibr" rid="B12">Charbuty and Abdulazeez, 2021</xref>)</td>
<td valign="top" align="left">2021</td>
<td valign="top" align="left">Private data generation</td>
<td valign="top" align="left">DT</td>
<td valign="top" align="left">Comparison among the models</td>
</tr>
<tr>
<td valign="top" align="center">(<xref ref-type="bibr" rid="B57">Sujjaviriyasup and Pitiruek, 2013</xref>)</td>
<td valign="top" align="left">2013</td>
<td valign="top" align="left">Thailand&#x2019;s Pacific white shrimp export data</td>
<td valign="top" align="left">ARIMA and SVM</td>
<td valign="top" align="left">SVM (MAE 1504.52, RMSE 1978.79, and MAPE 11.22%)</td>
</tr>
<tr>
<td valign="top" align="center">(<xref ref-type="bibr" rid="B34">Lee and Moon, 2014</xref>)</td>
<td valign="top" align="left">2014</td>
<td valign="top" align="left">Yearly yield of apple</td>
<td valign="top" align="left">Kernel smoothing model</td>
<td valign="top" align="left">MAPE 5.7 and R2 is 1</td>
</tr>
<tr>
<td valign="top" align="center">Ours</td>
<td valign="top" align="left">2023</td>
<td valign="top" align="left">Self-generated dataset</td>
<td valign="top" align="left">KRR (proposed) and SVR, NB, RR, RF, and CB</td>
<td valign="top" align="left">KRR obtains highest results</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3">
<title>Proposed paradigm</title>
<p>Crop production prediction is a major concern for an agriculture-based country like Bangladesh because many prospective crops can be planted in a single season. Currently, the farmers choose the crops for plantation on their own knowledge, which might not be an effective prediction every time. Sometimes, it might give better production, and sometimes, not, which would then be very harmful to the economy of such an agriculture-based country. Moreover, the government necessitates predicting crop production to estimate crop amount for the upcoming year. We design an ensemble ML model to predict crop production based on the environment, cultivation area, and previous production parameters. We first gather real-world data records from different periods (1969&#x2013;2021) of the diverse areas of Bangladesh and then propose an ensemble ML learning approach, called KRR, to accurately predict crop production on the basis of the environmental condition after inquiring about the most popular classical ML models. Using our KRR, the farmers can choose the best crops for the plantation, and the government can better estimate crop production for the next year. Notice that we do not find any such work to predict crop production in the Bangladesh context. Note that we discuss with a number of agriculturists to sort out the environmental factors related to the production of crops in Bangladesh. After that, we consider eight factors for predicting crop production, as illustrated in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. The final dataset contains 7,000 samples of five categories of crops (Aus rice, Aman rice, Boro rice, potato, and wheat), each having 1,400 samples. If we want to add other crops in this system, then the same dataset should be generated and then we need to train the best-performing model as the procedure of ML training and testing.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Short description of the attributes in the raw data records.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">No.</th>
<th valign="top" align="left">Attribute</th>
<th valign="top" align="left">Short Description</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">1</td>
<td valign="top" align="left">Rainfall</td>
<td valign="top" align="left">Average rainfall of the months responsible for the specific crop</td>
</tr>
<tr>
<td valign="top" align="left">2.</td>
<td valign="top" align="left">Maximum Temperature</td>
<td valign="top" align="left">Average maximum temperature of the months responsible for the specific crop</td>
</tr>
<tr>
<td valign="top" align="left">3</td>
<td valign="top" align="left">Minimum Temperature</td>
<td valign="top" align="left">Average minimum temperature of the months responsible for the specific crop</td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">Humidity</td>
<td valign="top" align="left">Average humidity of the months responsible for the specific crop</td>
</tr>
<tr>
<td valign="top" align="left">5</td>
<td valign="top" align="left">Wind Speed</td>
<td valign="top" align="left">Average wind speed of the months responsible for the specific crop</td>
</tr>
<tr>
<td valign="top" align="left">6</td>
<td valign="top" align="left">Cloud Coverage</td>
<td valign="top" align="left">Average cloud coverage of the months responsible for the specific crop</td>
</tr>
<tr>
<td valign="top" align="left">7</td>
<td valign="top" align="left">Bright Sunshine</td>
<td valign="top" align="left">Average bright sunshine of the months responsible for the specific crop</td>
</tr>
<tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">Aus Area</td>
<td valign="top" align="left">Total area of Aus cultivation including local area and High Yielding Variety (HYV) area in acres</td>
</tr>
<tr>
<td valign="top" align="left">9</td>
<td valign="top" align="left">Aman Area</td>
<td valign="top" align="left">Total area of Aman cultivation including local area and HYV area in acres</td>
</tr>
<tr>
<td valign="top" align="left">10</td>
<td valign="top" align="left">Boro Area</td>
<td valign="top" align="left">Total area of Boro cultivation including local area and HYV area in acres</td>
</tr>
<tr>
<td valign="top" align="left">11</td>
<td valign="top" align="left">Potato Area</td>
<td valign="top" align="left">Total area of potato cultivation including local area and HYV area in acres</td>
</tr>
<tr>
<td valign="top" align="left">12</td>
<td valign="top" align="left">Wheat Area</td>
<td valign="top" align="left">Total area of wheat cultivation including local area and HYV area in acres</td>
</tr>
<tr>
<td valign="top" align="left">13</td>
<td valign="top" align="left">Aus Production</td>
<td valign="top" align="left">Total production of Aus in tons</td>
</tr>
<tr>
<td valign="top" align="left">14</td>
<td valign="top" align="left">Aman Production</td>
<td valign="top" align="left">Total production of Aman in tons</td>
</tr>
<tr>
<td valign="top" align="left">15</td>
<td valign="top" align="left">Boro Production</td>
<td valign="top" align="left">Total production of Boro in tons</td>
</tr>
<tr>
<td valign="top" align="left">16</td>
<td valign="top" align="left">Potato Production</td>
<td valign="top" align="left">Total production of potato in tons</td>
</tr>
<tr>
<td valign="top" align="left">17</td>
<td valign="top" align="left">Wheat Production</td>
<td valign="top" align="left">Total production of wheat in tons</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3_1">
<title>Approach overview</title>
<p>We illustrate the working steps of the proposed crop production and recommendation paradigm in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. The first stage is dataset preparation, which delivers a suitable data format for training and testing using the proposed ensemble learning and the existing investigated ML approaches after the necessary preprocessing and feature extraction. After that, the evaluation and analysis are performed based on the experimental results. Finally, the recommended system is presented for suggesting suitable crops for cultivating a specific region in the next season.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Proposed methodology for predicting crop production. Raw data collection, dataset preparation, data preprocessing, model development, crop production prediction, and model evaluation with the significant test are all carried out in synchronization throughout the entire methodology.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1234555-g001.tif"/>
</fig>
<p>Top-down crop production prediction is depicted in the proposed methodology shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. After collecting raw data, we structure it crop-by-crop for the five regular crops in our research, jointly with the environmental variables during each of the crops&#x2019; relevant months. To create a dataset suitable for ML training, we handle missing values, mitigate wrong format and wrong data and modify raw data as required. In keeping with the standard ML practice, we split the final dataset into training and testing segments for the purposes of model training and testing with various evaluation methods. The ML models are trained independently using 80% of the training data and evaluated using the remaining 20%. After training and testing several models, we tabulate the evaluation outcome in terms of MSE, RMSE, MAE, and <italic>R</italic>
<sup>2</sup>. To determine the superiority of the proposed ensemble KRR, we conduct a DM significant test to compare it to the state-of-the-art benchmark ML models and determine its relative performance.</p>
</sec>
<sec id="s3_2">
<title>Dataset preparation</title>
<p>We collect the raw data samples from four different agricultural organizations in Bangladesh, which are BMD, BADC, BRRI, and BBS from 1969 to 2021 of different seasons, as shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>. Kharif and Rabi are the two harvest seasons in Bangladesh for the majority of crops. The environment varies depending on the harvest season. The months responsible for crop production are considered when constructing the dataset for each crop. The specific crop&#x2019;s weather information for the corresponding month is then provided. For example, Aus rice is harvested during the Kharif season, from June through August. It indicates that the monthly environmental data are regarded as weather data for Aus rice. Other crops&#x2019; samples are generated using the same manner, and the months corresponding to each season are listed in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>. In the final dataset, the data samples include 7,000 records of five categories of crops (Aus rice, Aman rice, Boro rice, potato, and wheat), each having 1,400 samples of different districts of Bangladesh. In particular, we prepare eight attributes (rainfall, maximum temperature, minimum temperature, humidity, wind speed, cloud coverage, bright sunshine, production area, and production amount) from a total of 17 original attributes to predict the production of a particular crop, as shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. As the weather of the different crops is different for the month, we take the average of maximum temperature, minimum temperature, rainfall, humidity, wind speed, cloud coverage, and bright sunshine for each crop according to the month.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Dataset description for different crops and environmental variables according to the season of the crops.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">No.</th>
<th valign="top" align="center">Crop Name</th>
<th valign="top" align="center">Harvest Season</th>
<th valign="top" align="center">Data Duration</th>
<th valign="top" align="center">Weather Data</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">1</td>
<td valign="top" align="center">Aus rice</td>
<td valign="top" align="center">Kharif</td>
<td valign="top" align="center">1969 to 2021</td>
<td valign="top" align="center">June to August</td>
</tr>
<tr>
<td valign="top" align="center">2</td>
<td valign="top" align="center">Aman rice</td>
<td valign="top" align="center">Rabi</td>
<td valign="top" align="center">1969 to 2021</td>
<td valign="top" align="left">December to January</td>
</tr>
<tr>
<td valign="top" align="center">3</td>
<td valign="top" align="center">Boro rice</td>
<td valign="top" align="center">Kharif</td>
<td valign="top" align="center">1969 to 2021</td>
<td valign="top" align="center">March to May</td>
</tr>
<tr>
<td valign="top" align="center">4</td>
<td valign="top" align="center">Potato</td>
<td valign="top" align="center">Kharif</td>
<td valign="top" align="center">1969 to 2021</td>
<td valign="top" align="center">February to March</td>
</tr>
<tr>
<td valign="top" align="center">5</td>
<td valign="top" align="center">Wheat</td>
<td valign="top" align="center">Rabi</td>
<td valign="top" align="center">1969 to 2021</td>
<td valign="top" align="left">November to March</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_3">
<title>Learning and evaluation</title>
<p>After generating the machine-learnable dataset, we split the dataset into training and testing sets. Then, we build the proposed ensemble ML (KRR) and investigated ML models (SVR, NB, CB, RF, and RR) using the training dataset and evaluate the trained models on the testing dataset to assemble the results. We consider four state-of-the-art performance indicators, i.e., mean absolute error (MAE), mean square error (MSE), root MSE (RMSE), and <italic>R</italic>
<sup>2</sup> score to evaluate the proposed and investigated ML models. All the experiments (dataset preprocessing, model training, model testing, and result processing) are accomplished using the Python programming language.</p>
</sec>
</sec>
<sec id="s4">
<title>Methods and measurements</title>
<sec id="s4_1">
<title>Data preprocessing</title>
<p>We preprocess the collected raw data to make them machine-learnable datasets. As all of the values of our dataset are numeric, we need not label any data during data preprocessing. Besides this, we normalize the data lastly to make it more trainable.</p>
<p>The preprocessing steps are as follows.</p>
<sec id="s4_1_1">
<title>Data cleaning</title>
<p>This step involves missing value handling, formatting, and wrong data handling. We handle missing values by replacing them with the mean of a feature, as illustrated in <xref ref-type="statement" rid="st1">Algorithm 1</xref>. Because crop production of a country is a continuous process, environmental variable values follow a pattern. Missing values can be the mean of the previous and next values in our dataset. For particular features, we format all the data in a unique form, which helps improve the performance of the ML models (<xref ref-type="bibr" rid="B56">Stekhoven and Buhlmann, 2012</xref>).</p>
</sec>
<sec id="s4_1_2">
<title>Data integration</title>
<p>We consider three types of data, i.e., environmental parameters related to crop production, areas of cultivation, and crop production amount of a particular area. We collect these data from different government organizations. To prepare the learnable dataset, we integrate all data into a single dataset.</p>
</sec>
<sec id="s4_1_3">
<title>Data reduction</title>
<p>Unnecessary, duplicate, and junk data are harmful to the performance of the ML models (<xref ref-type="bibr" rid="B47">Royston et&#xa0;al., 2006</xref>; <xref ref-type="bibr" rid="B5">Benjelloun et&#xa0;al., 2007</xref>). To make the best ML learnable dataset, we remove unnecessary, duplicates, and junk values.</p>
<statement id="st1">
<label>Algorithm 1 Missing value handling</label>
<p>
<preformat>
<bold>Input:</bold> Raw data (<bold><italic>S</italic></bold>)
<bold>Output:</bold> Preprocessed dataset after missing value handling
1: <bold>procedure</bold> <italic>MissingValueHandling</italic>(<bold><italic>S</italic></bold>)
2: &#x2003;<bold>for</bold> each attribute <italic>S<sup>a</sup></italic> <bold>do</bold>
3: <italic>&#x2003;m<sup>a</sup></italic> = mean (<italic>S<sup>a</sup></italic>) [<italic>m<sup>a</sup></italic> is the arithmetic mean of attribute <italic>S<sup>a</sup>]</italic>
4: &#x2003;<bold>for</bold> each sample data <italic>S<sub>d</sub><sup>a</sup></italic> <bold>do</bold>
5: &#x2003;<bold>if</bold> <italic>S<sub>d</sub><sup>a</sup></italic> is missing <bold>then</bold>
6: &#x2003;<italic>S<sub>d</sub><sup>a</sup></italic>:=<italic>m<sup>a</sup></italic>
7: <bold>&#x2003;end if</bold>
8: <bold>&#x2003;end for</bold>
9: &#x2003;<bold>end for</bold>
10: <bold>end procedure</bold>
</preformat>
</p>
</statement>
</sec>
<sec id="s4_1_4">
<title>Data normalization</title>
<p>Finally, we normalize the entire dataset to integer values to fit into the ML algorithms. We employ the classical min&#x2013;max normalization technique to normalize the dataset.</p>
</sec>
</sec>
<sec id="s4_2">
<title>Training models</title>
<p>In this section, we recapitulate the working principles of the investigated ML and ensemble ML approaches (SVR, NB, RR, RF, and CB) and present our proposed ensemble ML paradigm (KRR). We select five benchmark ML models instead of all ML algorithms in a strategy. ML algorithms can be classified based on architecture and working procedure. We choose SVM as the representative of the distance-based ML algorithm, RR from the group of regularization ML algorithms, RF as the representative of the bagging ensemble algorithm, NB as the member of the Bayes theorem means probability-based ML algorithm, and CB as the representative of boosting ensemble algorithm. We select those five algorithms to represent all ML algorithms in our analysis, train them individually using our dataset, and measure their performances to compare with the proposed ensemble KRR.</p>
<sec id="s4_2_1">
<title>SVR</title>
<p>SVR is a supervised ML algorithm that is a useful technique for both data classification and regression (<xref ref-type="bibr" rid="B55">Somvanshi et&#xa0;al., 2016</xref>). In regression, the data are separated into training and testing sets. Each instance in the training set contains one target value (class label) and several attributes named as the features (observed variables). The goal of SVR is to produce a model (based on the training data) that predicts the target values of the test data given only the test data attributes (<xref ref-type="bibr" rid="B51">Shang et&#xa0;al., 2016</xref>). According to the characteristics of our dataset, we use the linear SVR approach for predicting different agricultural crop production rates.</p>
</sec>
<sec id="s4_2_2">
<title>NB</title>
<p>NB is one of the most efficient and effective inductive ML algorithms (<xref ref-type="bibr" rid="B45">Ratanamahatana and Gunopulos, 2003</xref>) (<xref ref-type="bibr" rid="B40">Panda and Patra, 2008</xref>). This uses the Bayes theorem to calculate the probability and then form a prediction. The basic insight of Bayes&#x2019; theorem is that when new data are introduced, the probability of an event may be changed. The NB model is simple to implement and it does not require sophisticated iterative parameter estimation, making it perfect for large datasets (<xref ref-type="bibr" rid="B46">Razzaghi et&#xa0;al., 2016</xref>).</p>
</sec>
<sec id="s4_2_3">
<title>RR</title>
<p>RR is a model optimization technique (<xref ref-type="bibr" rid="B58">Tavares et&#xa0;al., 2021</xref>). It estimates the coefficients of multiple regression models under conditions of high correlation between linearly independent variables. This model is also known as a regularization model and uses the <italic>L</italic>2 regularization process. It has been applied in various disciplines, including agricultural data, engineering, chemistry, and econometrics. RR creates a new matrix by adding a ridge parameter (<italic>k</italic>) from the identity matrix to the cross-product matrix. The reason it is known as <italic>ridge regression</italic> is that the correlation matrix&#x2019;s diagonal of one can be compared to a ridge. Overfitting is a problem that RR solves since squared error regression by itself can distinguish between significant and insignificant features, using all of them instead, resulting in overfitting (<xref ref-type="bibr" rid="B21">Garriga et&#xa0;al., 2017</xref>). RR introduces a small amount of bias in order to match the model to the actual values of the data. However, it does not have the ability to do feature selection and the final model includes all predictors. It swaps variance for bias and decreases coefficients toward zero.</p>
</sec>
<sec id="s4_2_4">
<title>RF</title>
<p>RF is an ensemble ML classifier that uses randomness to create a group of independent and non-identical DTs (<xref ref-type="bibr" rid="B44">Provost et&#xa0;al., 2016</xref>). This algorithm is used for both classification and regression purposes and it is a combination of tree predictors. Each DT has a random vector as a parameter, determines the feature of samples at random, and selects the training dataset from either a subset of the dataset at random (<xref ref-type="bibr" rid="B8">Bradter et&#xa0;al., 2013</xref>). Whenever a random selection of features is used to split each node, the error rates are equivalent to Ad boost, but they are more robust in terms of turbulence (<xref ref-type="bibr" rid="B50">Shakoor et&#xa0;al., 2017</xref>). RF is a highly flexible and easy-to-use ML algorithm that produces, even without hyper-parameter tuning, a great result most of the time. In this work, we use RF for the regression aspect of this algorithm based on our necessity. We successfully achieve a very high accuracy upon implementation of our dataset using this RF regression. Python&#x2019;s scikit-learn has a helpful tool for this that quantifies the relevance of a feature by looking at how much inaccuracy is minimized across all trees in the forest by tree nodes using that feature (<xref ref-type="bibr" rid="B24">Grange and Hand, 1987</xref>). Deep DTs might suffer from overfitting but RF prevents overfitting most of the time. It creates random subsets of the features and builds smaller trees using these subsets, and afterward, it combines the sub-trees.</p>
</sec>
<sec id="s4_2_5">
<title>CB</title>
<p>CB is an ML algorithm for gradient boosting on DTs. Gradient boosted DTs are a powerful tool for classification and regression. This algorithm is developed by Yandex researchers and engineers and it is the successor of the MatrixNet algorithm. It is widely used for ranking tasks, forecasting, and making recommendations (<xref ref-type="bibr" rid="B25">Hancock and Khoshgoftaar, 2020</xref>). This supervised algorithm is used both for classification and regression purposes. CB is a special type of boosting algorithm with much less prediction time for its symmetric tree structure. However, it is sensitive to its hyperparameter tuning.</p>
</sec>
<sec id="s4_2_6">
<title>The proposed ensemble ML approach</title>
<p>The main purpose of introducing an ensemble regressor is to reduce the variance of the data during model training (<xref ref-type="bibr" rid="B48">Sagi and Rokach, 2018</xref>). It helps to fit the data to the models, and the model can predict more accurately. In the proposed KRR ensemble method, we use a distance-based algorithm KNN, a regularization method RR, and a tree-based ensemble RF. The KNN model is simple to implement and works well with non-linear data. Because it does not require calculating any fixed parameters or values, fitting the model also takes little time. The KNN algorithm makes predictions about the significance of new data points based on their &#x201c;feature similarity&#x201d;. A score is given to the new point based on how similar it is to the points used for training. RR is good for preventing overfitting, which adds one additional element to the cost function of linear regression. The primary reason these penalty terms are included is to ensure regularization, or the reduction of model weights to zero or close to zero so that the model does not overfit the data. Nonlinear parameters do not affect the performance of an RF, unlike curve-based techniques. As a result, if the non-linearity between the independent variables is high, RF may beat other curve-based methods. It is usually robust to outliers and can handle them automatically. It does not require feature scaling (standardization and normalization) because it employs a rule-based method rather than distance calculation. That is the reason for creating the new ensemble model using the algorithms that can handle overfitting by themselves, with no need for extra preprocessing when needed during training and testing. A second-order ensemble strategy called blending is used to construct this KRR regression method. Blending ensemble ML methods find the best combination of the predictors from the three ML algorithms (KNN, RR, and RF) and form an ensemble regressor for better prediction (<xref ref-type="bibr" rid="B20">Farooq et&#xa0;al., 2021</xref>). The blending process is the same as the stacking ensemble procedure, but it has some unique differences. Stacking uses out-of-fold prediction for the training set of the next layer in the meta-model. On the other hand, our blending uses a validation set (10%&#x2013;15% of the training data) to train the next layer in the meta-model. KRR combines the mapping functions learned by the contributing members. Our proposed KRR is the combination of the hyperplanes of KNN, RF, and RR. The working procedure and function mapping of KRR are shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Block diagram of the proposed KRR approach. KRR is built with the three mostly used ML algorithms: KNN, RF, and RR, using the blending ensemble strategy.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1234555-g002.tif"/>
</fig>
<p>The working principle of KRR is different from the KNN, RF, and RR, which are the building blocks of the ensemble approach. The proposed KRR and RF are both ensemble methods, where KRR is a blending ensemble where the building blocks work individually to find the main output. RF is a bagging ensemble where data are mainly divided into several bags to train the individual tress to formulate the result. The benchmark models sometimes fall into overfitting problems, and due to the high variance of the data points, the performance of the models falls in some cases, but the proposed method outperforms in this case by its tolerance and flexibility of learning from the dataset.</p>
<p>Our deployed KRR can be a solution to this type of regression issue for better performance and the best fitting of datasets. The criteria for adopting the proposed scheme KRR to another dataset are very easy. As with the traditional ML training and testing process, the training data must fit the KRR architecture and then evaluated by the remaining testing data. The KRR architecture is already described above. However, hyperparameter tuning of the building block ML algorithms of KRR can bring a better result when adopted with other datasets.</p>
<sec id="s4_2_6_1">
<title>Complexity of proposed ensemble KRR</title>
<p>The complexity of KRR can be written into two steps. In the first step, the complexity of stacking architecture forms and then the individual algorithm&#x2019;s complexity is added one by one in the next step as follows: The complexity of the first step is O(B(C + R)), where R represents the number of replacements, the number of bags of the dataset is B, and C represents the number of classifiers in the ensemble algorithm. In the second step:</p>
<list list-type="simple">
<list-item>
<p>1. Complexity KNN is O(nd), where n is the number of training examples, and d is the number of features.</p>
</list-item>
<list-item>
<p>2. Complexity of RR is O(<italic>n</italic>
<sup>3</sup>), where n is the number of data.</p>
</list-item>
<list-item>
<p>3. Random Forest of size T and maximum depth D (excluding the root) is O(T.D).</p>
</list-item>
</list>
</sec>
</sec>
</sec>
<sec id="s4_3">
<title>Model evaluation</title>
<p>The classical and ensemble ML algorithms and our proposed ensemble ML scheme are applied to predict crop production. The training data train these approaches, and the model learns the data sequences and then forms a prediction. The performance of the ML models is calculated using four evaluation metrics, i.e., MAE, MSE, RMSE, and <italic>R</italic>
<sup>2</sup>. MSE can be defined as the absolute value of the difference between the predicted and actual value. Using MSE in regression will penalize large errors more than small ones if we assume that the target follows a normal distribution. The MSE is calculated as:</p>
<disp-formula>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>MAE indicates how big of an error we may expect on average from the prediction (<xref ref-type="bibr" rid="B38">Morales and Villalobos, 2023</xref>). MSE indicates how close it is to a set of points. It accomplishes this by squaring the distances between the points and the regression line (these distances are the errors). Squaring is required to eliminate any negative signs. The MAE is calculated as:</p>
<disp-formula>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>RMSE is the standard deviation of the residuals (prediction errors) (<xref ref-type="bibr" rid="B22">Glennie and Lichti, 2010</xref>). Residuals are a measure of how far the data points are from the regression line; RMSE is a measure of how to spread out these residuals. In other words, it reveals how strongly the data are aggregated around the line of best fit. The RMSE is calculated as:</p>
<disp-formula>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<italic>R</italic>
<sup>2</sup> is a statistical measure of how much variation in a dependent variable can be explained by variation in the independent variables. The main objective of this score is to predict future results based on existing data. The extent to which the model can reproduce observed results is quantified by this measure, which is based on the fraction of the total variation in outcomes that can be attributed to the model. <italic>R</italic>
<sup>2</sup> is calculated as:</p>
<disp-formula>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>Y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>Y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the above equations, the <italic>Y<sub>i</sub></italic>indicates the actual value, <italic>Y</italic>
<sup>&#x2c6;</sup>
<italic>
<sub>i</sub></italic>indicates the predicted value, <italic>Y</italic>&#xaf; indicates the means of the <italic>Y</italic> values, and <italic>n</italic> is the total number of samples.</p>
</sec>
</sec>
<sec id="s5">
<title>Experiment and result analysis</title>
<sec id="s5_1">
<title>Experimental setting</title>
<p>We use Python&#x2019;s <italic>scikit-learn</italic> tool to construct the proposed ensemble ML scheme (KRR) as well as the investigated classical and ensemble ML models (SVR, NB, RR, RF, and CB). We consider the actual vs. predicted curve and error metrics (MAE, MSE, RMSE, and <italic>R</italic>
<sup>2</sup>) as the evaluation parameters of the trained ML models. We use supervised methods to predict crop production, where the dataset contains eight features and the target is the amount of crop production in a certain area. We take the average results of experiments in three phases, such as 80:20, 50:50, and 30:70 training and testing ratio, where each phase has 10 trials.</p>
<p>To get better performance, we tune the hyperparameters of the proposed ensemble ML scheme as well as the investigated classical and ensemble ML models. The same hyperparameters give a better result for almost all experiments. In particular, SVR gives better results with the linear kernel when <italic>c</italic> = 100 and gamma is <italic>auto</italic> while we use 10-fold cross-validation to find the value of <italic>gamma</italic> and <italic>c</italic>. Gaussian NB achieves a better result in all cases with the hyperparameters i.e., <italic>estimator</italic> = <italic>model</italic>, <italic>param_grid</italic> = <italic>params_nb</italic>, and <italic>cv</italic> = <italic>cv_methods</italic>. RF finds <italic>n_estimator</italic> = 20, and <italic>random_state</italic> = 42 in all cases for better performance. For all experiments, RR gives maximum performance when <italic>alpha</italic> = 0.01. CB model gives a better result when <italic>estimator</italic> = <italic>model_cvr</italic>, <italic>cv</italic> = 2 n_jobs = &#x2212;1, and <italic>learningrate</italic> = 0.05. Our proposed model KRR achieves high accuracy with a low error rate with the hyperparameters <italic>alpha</italic> = 0.01, <italic>n_estimator</italic> = 10, and <italic>random_state</italic> = 42 for almost all experiments.</p>
</sec>
<sec id="s5_2" sec-type="results">
<title>Result analysis on Aus rice production</title>
<p>Aus is considered one of the major crops in Bangladesh. This type of rice is closely related to indica-type rice but it has a distinct genetic group (<xref ref-type="bibr" rid="B10">Chakravarthi and Naravaneni, 2006</xref>). Still, this variety is cultivated under environmental stress conditions in Bangladesh and India (<xref ref-type="bibr" rid="B6">Berger et&#xa0;al., 2004</xref>). The value of Aus production varies according to the environment and the region of cultivation. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref> represents the actual vs. predicted rice production in every fiscal year from 2015 to 2021 in the Dinajpur zone of Bangladesh. The <italic>x</italic>-axis symbolizes the fiscal year, while the <italic>y</italic>-axis reflects rice production (both actual and predicted). It clearly indicates that our proposed algorithm outperforms the other traditional ML and ensemble ML algorithms. We also evaluate MAE, MSE, RMSE, and <italic>R</italic>
<sup>2</sup> to measure the model&#x2019;s goodness of fit in predicting rice production in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>, which demonstrates that the KRR model fits better than the other models we investigate. <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> illustrates that KKR has less mistakes, such as 9.11% MAE, approximately 1% MSE, and 9.17% RMSE, than the others.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Comparison of the actual and predicted values of Aus rice production and error rating of the investigated and proposed ML models (SVR, NB, RF, RR, CB, and KRR) in each fiscal year from 2015 to 2021. <bold>(A)</bold> Actual vs Predicted bar chart for Aus. <bold>(B)</bold> Line chart for the error rating of Aus production prediction.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1234555-g003.tif"/>
</fig>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Error ratings using the investigated and proposed ML approaches for predicting Aus rice production using different metrics and <italic>R</italic>
<sup>2</sup> score.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">MAE</th>
<th valign="top" align="center">MSE</th>
<th valign="top" align="center">RMSE</th>
<th valign="top" align="left">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVR</td>
<td valign="top" align="left">0.107</td>
<td valign="top" align="left">0.016</td>
<td valign="top" align="left">0.125</td>
<td valign="top" align="left">0.980</td>
</tr>
<tr>
<td valign="top" align="left">NB</td>
<td valign="top" align="left">0.104</td>
<td valign="top" align="left">0.019</td>
<td valign="top" align="left">0.136</td>
<td valign="top" align="left">0.980</td>
</tr>
<tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">0.098</td>
<td valign="top" align="left">0.010</td>
<td valign="top" align="left">0.101</td>
<td valign="top" align="left">0.990</td>
</tr>
<tr>
<td valign="top" align="left">RR</td>
<td valign="top" align="left">0.091</td>
<td valign="top" align="left">0.018</td>
<td valign="top" align="left">0.135</td>
<td valign="top" align="left">0.980</td>
</tr>
<tr>
<td valign="top" align="left">CB</td>
<td valign="top" align="left">0.115</td>
<td valign="top" align="left">0.026</td>
<td valign="top" align="left">0.161</td>
<td valign="top" align="left">0.970</td>
</tr>
<tr>
<td valign="top" align="left">KRR</td>
<td valign="top" align="left">0.091</td>
<td valign="top" align="left">0.009</td>
<td valign="top" align="left">0.099</td>
<td valign="top" align="left">0.990</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>, it can also be observed that there is no linear relationship between the area and environmental data, and Aus rice production. The weather conditions in Bangladesh vary from year to year, and natural disasters may occur. In August 2017, during the Kharif harvest season, an uncertain flood occurred in Dinajpur zone (<xref ref-type="bibr" rid="B17">Das and Rahman, 2018</xref>). It damaged the crops and interrupted the production cycle. Our proposed KRR performs effectively during this period, which demonstrates its adaptability to the uncertainty in the environmental data. Owing to a lack of soil fertility and improper management of soil carbon, the post-flood effects on Aus production continue throughout the subsequent growing seasons (<xref ref-type="bibr" rid="B54">Siddique et&#xa0;al., 2022</xref>). In this uncertain situation, the proposed KRR performs better than other models. This type of prediction is critical for farmers as well as individuals who depend on harvesting for a living. Such future production prediction aids in the care of alternative solutions to ensure food and industrial raw materials.</p>
</sec>
<sec id="s5_3" sec-type="results">
<title>Result analysis on Aman rice production</title>
<p>Aman rice is grown in Bangladesh during the winter (rabi) season. The cultivation of Aman rice is strongly linked to the environment. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref> illustrates the actual and predicted values using the investigated and proposed ML algorithms. The performance of our proposed algorithm KRR reaches maximum accuracy for each fiscal year. In terms of error measurement parameters, both our proposed KRR and the RF models have the same <italic>R</italic>
<sup>2</sup> score. However, our KRR obtains better MAE, MSE, and RMSE than RF and other models, as shown in <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref> indicates that our KRR is the best-fit model compared to others. Changes in maximum temperatures have had a significant impact on crop yield in Bangladesh. However, temperature changes confirm that maximum temperature raises the risk for Aman rice while minimum temperature reduces yield variability. Rainfall has increased the risk of Aman rice (<xref ref-type="bibr" rid="B49">Sarker et&#xa0;al., 2019</xref>). The great news is that environmental factors in Bangladesh are now changing within a range that allows Aman rice to adapt to the environment. In recent years, Aman rice has been consistently produced because of its versatility (<xref ref-type="bibr" rid="B11">Chakrobarty et&#xa0;al., 2021</xref>). <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref> shows the consistency of Aman rice production, and in most of the cases, the proposed KRR performs better. The authority can benefit from this method in their long and short plan for food supply in the future.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Comparison of the actual and predicted values of Aman rice production and error rating of the investigated and proposed ML models (SVR, NB, RF, RR, CB, and KRR) in each fiscal year from 2015 to 2021. <bold>(A)</bold> Actual vs Predicted bar chart for Aman. <bold>(B)</bold> Line chart for the error rating of Aman production prediction.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1234555-g004.tif"/>
</fig>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Error ratings using the investigated and proposed ML approaches for predicting Aman rice production using different metrics and <italic>R</italic>
<sup>2</sup> score.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">MAE</th>
<th valign="top" align="center">MSE</th>
<th valign="top" align="center">RMSE</th>
<th valign="top" align="left">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVR</td>
<td valign="top" align="left">1.055</td>
<td valign="top" align="left">1.510</td>
<td valign="top" align="left">1.233</td>
<td valign="top" align="left">0.790</td>
</tr>
<tr>
<td valign="top" align="left">NB</td>
<td valign="top" align="left">0.757</td>
<td valign="top" align="left">1.326</td>
<td valign="top" align="left">1.152</td>
<td valign="top" align="left">0.860</td>
</tr>
<tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">0.736</td>
<td valign="top" align="left">0.962</td>
<td valign="top" align="left">0.981</td>
<td valign="top" align="left">0.900</td>
</tr>
<tr>
<td valign="top" align="left">RR</td>
<td valign="top" align="left">0.759</td>
<td valign="top" align="left">1.050</td>
<td valign="top" align="left">1.025</td>
<td valign="top" align="left">0.890</td>
</tr>
<tr>
<td valign="top" align="left">CB</td>
<td valign="top" align="left">0.778</td>
<td valign="top" align="left">1.196</td>
<td valign="top" align="left">1.093</td>
<td valign="top" align="left">0.870</td>
</tr>
<tr>
<td valign="top" align="left">KRR</td>
<td valign="top" align="left">0.709</td>
<td valign="top" align="left">0.921</td>
<td valign="top" align="left">0.959</td>
<td valign="top" align="left">0.900</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5_4" sec-type="results">
<title>Result analysis on Boro rice production</title>
<p>Boro rice is cultivated in the Kharif season, which has a vital impact on the total rice production in Bangladesh. <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5A</bold>
</xref> illustrates the actual vs. predicted bar chart for the Boro production from 2015 to 2021, while <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5B</bold>
</xref> indicates that KRR is the best-fit model compared to others. <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref> indicates that the performance using RF is better than KRR in respect of MAE and MSE. However, RMSE and <italic>R</italic>
<sup>2</sup> are good in KRR. In summary, the average performance of KRR is better than the other models.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Comparison of the actual and predicted values of Boro rice production and error rating of the investigated and proposed ML models (SVR, NB, RF, RR, CB, and KRR) in each fiscal year from 2015 to 2021. <bold>(A)</bold> Actual vs Predicted bar chart for Boro. <bold>(B)</bold> Line chart for the error rating of Boro production prediction.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1234555-g005.tif"/>
</fig>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Error rating representation using the investigated and proposed ML approaches for predicting Boro rice production using different metrics and <italic>R</italic>
<sup>2</sup> score.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">MAE</th>
<th valign="top" align="center">MSE</th>
<th valign="top" align="center">RMSE</th>
<th valign="top" align="left">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVR</td>
<td valign="top" align="left">0.535</td>
<td valign="top" align="left">0.553</td>
<td valign="top" align="left">0.744</td>
<td valign="top" align="left">0.960</td>
</tr>
<tr>
<td valign="top" align="left">NB</td>
<td valign="top" align="left">0.489</td>
<td valign="top" align="left">0.478</td>
<td valign="top" align="left">0.422</td>
<td valign="top" align="left">0.960</td>
</tr>
<tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">0.285</td>
<td valign="top" align="left">0.222</td>
<td valign="top" align="left">0.471</td>
<td valign="top" align="left">0.980</td>
</tr>
<tr>
<td valign="top" align="left">RR</td>
<td valign="top" align="left">0.447</td>
<td valign="top" align="left">0.312</td>
<td valign="top" align="left">0.559</td>
<td valign="top" align="left">0.970</td>
</tr>
<tr>
<td valign="top" align="left">CB</td>
<td valign="top" align="left">0.453</td>
<td valign="top" align="left">0.259</td>
<td valign="top" align="left">0.399</td>
<td valign="top" align="left">0.980</td>
</tr>
<tr>
<td valign="top" align="left">KRR</td>
<td valign="top" align="left">0.446</td>
<td valign="top" align="left">0.246</td>
<td valign="top" align="left">0.376</td>
<td valign="top" align="left">0.990</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To this end, we investigate three major rice variations in this part to predict their production concerning the environmental conditions. Given the analyses, we can deduce that our proposed model KRR performs better in predicting different rice production in Bangladesh. Boro rice needs extra irrigation for cultivation. The average production of Boro rice is expected to decrease by over 20% in 2050 and by 50% in 2070 as a result of climate change (<xref ref-type="bibr" rid="B4">Basak et&#xa0;al., 2010</xref>). It has been determined that an increase in both the maximum and minimum temperatures is the primary cause of a reduction in yield. Rainfall pattern changes during the growing season have also been observed to impact rice production and irrigation needs. Using the proposed KRR, researchers can track environmental factor changes and then take the necessary steps to select an alternative rice variety or predict the production of the new variety.</p>
</sec>
<sec id="s5_5" sec-type="results">
<title>Result analysis on potato production</title>
<p>In Bangladesh, potato farming takes place throughout the winter season. Sandy loam soils can produce more potatoes than other types of soil (<xref ref-type="bibr" rid="B19">Faraji et&#xa0;al., 2017</xref>). In terms of productivity and internal demand, potatoes are a popular crop in Bangladesh. As a result, predicting potato production has a significant influence on the economy. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref> demonstrates a comparison among the actual and predicted values using the investigated and proposed ML approaches, which shows that our proposed KRR approach offers superior prediction in almost all cases. According to <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref>, our proposed approach KRR outperforms the other investigated ML algorithms in all error measures. In particular, KRR predicts potato production with a minimum MSE of 6.3%. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref> illustrates that KRR is the best-fit model to predict potato production. Potato production in Bangladesh is still at a satisfactory level but it swings to change of environment (<xref ref-type="bibr" rid="B26">Hossain and Abdulla, 2016</xref>). In some consecutive years, the production goes down due to heavy cold and attack of unexpected diseases on potato (<xref ref-type="bibr" rid="B27">Islam et&#xa0;al., 2022</xref>). To predict production in this kind of uncertain situation, the proposed KRR can be a good solution for the agriculture domain people.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Comparison of the actual and predicted values of potato production and error rating of the investigated and proposed ML models (SVR, NB, RF, RR, CB, and KRR) in each fiscal year from 2015 to 2021. <bold>(A)</bold> Actual vs Predicted bar chart for Potato. <bold>(B)</bold> Line chart for the error rating of Potato production prediction.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1234555-g006.tif"/>
</fig>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Error ratings using the investigated and proposed ML approaches for predicting potato production using different metrics and <italic>R</italic>
<sup>2</sup> score.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">MAE</th>
<th valign="top" align="center">MSE</th>
<th valign="top" align="center">RMSE</th>
<th valign="top" align="left">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVR</td>
<td valign="top" align="left">0.416</td>
<td valign="top" align="left">0.474</td>
<td valign="top" align="left">0.688</td>
<td valign="top" align="left">0.960</td>
</tr>
<tr>
<td valign="top" align="left">NB</td>
<td valign="top" align="left">0.735</td>
<td valign="top" align="left">0.275</td>
<td valign="top" align="left">0.525</td>
<td valign="top" align="left">0.970</td>
</tr>
<tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">0.299</td>
<td valign="top" align="left">0.284</td>
<td valign="top" align="left">0.533</td>
<td valign="top" align="left">0.980</td>
</tr>
<tr>
<td valign="top" align="left">RR</td>
<td valign="top" align="left">0.735</td>
<td valign="top" align="left">0.873</td>
<td valign="top" align="left">0.934</td>
<td valign="top" align="left">0.930</td>
</tr>
<tr>
<td valign="top" align="left">CB</td>
<td valign="top" align="left">0.787</td>
<td valign="top" align="left">2.144</td>
<td valign="top" align="left">1.464</td>
<td valign="top" align="left">0.840</td>
</tr>
<tr>
<td valign="top" align="left">KRR</td>
<td valign="top" align="left">0.134</td>
<td valign="top" align="left">0.062</td>
<td valign="top" align="left">0.250</td>
<td valign="top" align="left">0.990</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5_6" sec-type="results">
<title>Result analysis on wheat production</title>
<p>In Bangladesh, the production of wheat is decreasing on average by 0.44% each year. People are cultivating different crops instead of wheat for more benefits and a higher production rate. The prediction of the production of wheat can improve the production rate of wheat. We use the same ML and ensemble ML algorithms to predict wheat production in the Dinajpur zone of Bangladesh.</p>
<p>
<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7A</bold>
</xref> indicates a comparison among the actual and predicted values of wheat rice production using the investigated (SVR, NB, RF, RR, and CB) and proposed (KRR) ML models in each fiscal year from 2015 to 2021. It demonstrates that the performance of our proposed KRR is better than other ML models. In terms of other error metrics, KRR achieves the best result than the other investigated ML approaches, as illustrated in <xref ref-type="table" rid="T9">
<bold>Table&#xa0;9</bold>
</xref> and <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7B</bold>
</xref>. Wheat production in South Asia climbed from 15 million tons in the 1960s to 95.5 million tons in 2004&#x2013;2005. It still needs to increase at a rate of 2%&#x2013;2.5% every year till the middle of the 21st century (<xref ref-type="bibr" rid="B13">Chatrath et&#xa0;al., 2007</xref>). Because there is little scope for growing wheat field areas, the main task will be to crack the yield barrier utilizing practical genetic and morphological techniques. Other issues are unique to the highly productive rice&#x2013;wheat farming system prevalent in the Indo-Gangetic plains. Though the production is at a low level, the 2017&#x2013;2078 and 2018&#x2013;2019 time periods have broken records. Previously, we discussed that the damage of the Aus rice due to flood plays a vital role in this segment (<xref ref-type="bibr" rid="B17">Das and Rahman, 2018</xref>). People engaged more in cultivating wheat to recover the damage to the economy in the period. It indicates that cultivating more wheat can be a solution to increase the amount of wheat production, which leads the agricultural economy in another direction.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Comparison of the actual and predicted values of wheat production and error rating of the investigated and proposed ML models (SVR, NB, RF, RR, CB, and KRR) in each fiscal year from 2015 to 2021. <bold>(A)</bold> Actual vs Predicted bar chart for Wheat. <bold>(B)</bold> Line chart for the error rating of Wheat production prediction.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1234555-g007.tif"/>
</fig>
<table-wrap id="T9" position="float">
<label>Table&#xa0;9</label>
<caption>
<p>Error ratings using the investigated and proposed ML approaches for predicting wheat production using different metrics and <italic>R</italic>
<sup>2</sup> score.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Model</th>
<th valign="top" align="center">MAE</th>
<th valign="top" align="center">MSE</th>
<th valign="top" align="center">RMSE</th>
<th valign="top" align="left">
<italic>R</italic>
<sup>2</sup>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVR</td>
<td valign="top" align="left">0.139</td>
<td valign="top" align="left">0.031</td>
<td valign="top" align="left">0.177</td>
<td valign="top" align="left">0.960</td>
</tr>
<tr>
<td valign="top" align="left">NB</td>
<td valign="top" align="left">0.086</td>
<td valign="top" align="left">0.019</td>
<td valign="top" align="left">0.137</td>
<td valign="top" align="left">0.970</td>
</tr>
<tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">0.067</td>
<td valign="top" align="left">0.019</td>
<td valign="top" align="left">0.099</td>
<td valign="top" align="left">0.980</td>
</tr>
<tr>
<td valign="top" align="left">RR</td>
<td valign="top" align="left">0.129</td>
<td valign="top" align="left">0.026</td>
<td valign="top" align="left">0.162</td>
<td valign="top" align="left">0.960</td>
</tr>
<tr>
<td valign="top" align="left">CB</td>
<td valign="top" align="left">0.079</td>
<td valign="top" align="left">0.017</td>
<td valign="top" align="left">0.129</td>
<td valign="top" align="left">0.980</td>
</tr>
<tr>
<td valign="top" align="left">KRR</td>
<td valign="top" align="left">0.023</td>
<td valign="top" align="left">0.016</td>
<td valign="top" align="left">0.013</td>
<td valign="top" align="left">0.990</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To summarize, we can state that, on average, our proposed model KRR outperforms the others in predicting crop production for all crops considered in this work. Regarding MSE, MAE, RMSE, and <italic>R</italic>
<sup>2</sup>, the proposed KRR performs better. From <xref ref-type="table" rid="T5">
<bold>Tables&#xa0;5</bold>
</xref> to <xref ref-type="table" rid="T9">
<bold>9</bold>
</xref>, we can clearly differentiate the performance of each model for prediction. In almost all cases, the MSE, MAE, and RMSE values of KRR are smaller than those of the other models, which indicates that KRR shows minimum error in the case of prediction compared to other ML models. However, the <italic>R</italic>
<sup>2</sup> value of the KRR is larger than the other models in the above-mentioned tables. It also creates a comparison among the models that KRR is a better fit to the dataset than existing benchmark ML models. To find the superiority of KNN, the DM significant test is also performed below.</p>
</sec>
<sec id="s5_7">
<title>Significant test on the superiority of the proposed ensemble ML model</title>
<sec id="s5_7_1">
<title>DM test</title>
<p>The DM test is one of the most used significant test procedures to compare the robustness of the best method in prediction. This is an asymptotic z-test of the hypothesis that calculates the loss difference (<xref ref-type="bibr" rid="B18">Diebold, 2015</xref>). In this study, we consider the null hypothesis as <italic>H0</italic>, i.e., the loss difference of model A is lower than or equal to that of model B. Note that the hypothesis rejection means model B is significantly more accurate than model A. In every hypothesis, testing model B is our proposed KRR ensemble model.</p>
</sec>
<sec id="s5_7_2">
<title>Significant test result</title>
<p>We use the DM test to find the significance of our proposed ensemble algorithm KRR compared to the other investigated ML models. We evaluate this test for the Aus rice, Aman rice, Boro rice, wheat, and potato data.</p>
<p>
<xref ref-type="table" rid="T10">
<bold>Table&#xa0;10</bold>
</xref> illustrates the DM values of different models compared to our proposed KRR. For almost all of the cases, our proposed KRR shows 5% significance over other ML models.</p>
<table-wrap id="T10" position="float">
<label>Table&#xa0;10</label>
<caption>
<p>DM value with significance for each investigated ML model compared to our proposed KRR ensemble model in terms of DM and <italic>p</italic>-values.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Investigated Model</th>
<th valign="top" align="left">Aus Rice (DM Value)</th>
<th valign="top" align="left">Aman Rice (DM Value)</th>
<th valign="top" align="left">Boro Rice (DM Value)</th>
<th valign="top" align="left">Potato (DM Value)</th>
<th valign="top" align="left">Wheat (DM Value)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SVR</td>
<td valign="top" align="left">23.041*</td>
<td valign="top" align="left">16.323*</td>
<td valign="top" align="left">17.554*</td>
<td valign="top" align="left">12.253*</td>
<td valign="top" align="left">13.862*</td>
</tr>
<tr>
<td valign="top" align="left">NB</td>
<td valign="top" align="left">22.060*</td>
<td valign="top" align="left">15.542*</td>
<td valign="top" align="left">5.884**</td>
<td valign="top" align="left">13.870*</td>
<td valign="top" align="left">11.960**</td>
</tr>
<tr>
<td valign="top" align="left">RF</td>
<td valign="top" align="left">2.561***</td>
<td valign="top" align="left">2.870***</td>
<td valign="top" align="left">22.021*</td>
<td valign="top" align="left">10.530**</td>
<td valign="top" align="left">9.532**</td>
</tr>
<tr>
<td valign="top" align="left">RR</td>
<td valign="top" align="left">4.292***</td>
<td valign="top" align="left">4.454***</td>
<td valign="top" align="left">0.460***</td>
<td valign="top" align="left">11.984**</td>
<td valign="top" align="left">12.933**</td>
</tr>
<tr>
<td valign="top" align="left">CB</td>
<td valign="top" align="left">6.160**</td>
<td valign="top" align="left">5.920**</td>
<td valign="top" align="left">14.744*</td>
<td valign="top" align="left">10.953**</td>
<td valign="top" align="left">8.192**</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Observation represents the algorithm&#x2019;s DM value of Aus rice, Aman rice, Boro rice, wheat, and potato while *, **, *** represent the significance level according to the p-values of the test. * represents 1% significance, ** represents 5% significance, and *** represents 10% significance of our proposed model against the investigated algorithms.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s6">
<title>Crop recommender system</title>
<p>Besides crop production forecasting, crop recommendation is a vital part of such a study. Suitable crops in suitable land can boost the production of any crop (<xref ref-type="bibr" rid="B7">Bhullar et&#xa0;al., 2023</xref>). Finding the best crops for the appropriate land is a challenging task. A complex analysis of the environmental variables and production rate is required to recommend a land for production (<xref ref-type="bibr" rid="B60">Van Ittersum et&#xa0;al., 2013</xref>). Every area has a unique value of environmental variables. Considering the standard variables as threshold values (collected from expert agriculturists), we propose to recommend crops for any specific area. We provide the block diagram in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>. First, the model is trained on the environmental and area data, and then, based on the environment and area data of the current year, the estimated production is predicted for each crop. Next, the production of the individual crops is compared with the threshold value of the crop for any specific area. If the production satisfies the criteria, then the crop will go to the list of recommendations. The threshold value is to be estimated by the associated local authority, which can be changed according to changes in the status of the area, demand, policy, and also environmental conditions. Comparing the crops of the same harvesting period, the model will recommend the right crops for the right place. We also introduce the pseudocode to recommend the crops for the land in <xref ref-type="statement" rid="st2">Algorithm 2</xref>. We now discuss every step in the recommender system as follows.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Block diagram of the proposed crop recommender system. This recommender system employs our proposed pre-trained KRR ML model to predict the production values of the test samples of different crops. After that, a suitable crop to grow is recommended using the predicted production values and the expert-defined threshold.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-14-1234555-g008.tif"/>
</fig>
<sec id="s6_1">
<title>Dataset creation</title>
<p>Dataset is one of the significant components of the recommendation system (<xref ref-type="bibr" rid="B64">Zhang et&#xa0;al., 2018</xref>). A dataset should train the model used for prediction and recommendation to understand the nature of the system. Then, it can predict and recommend the outcome. In this recommendation system, environmental, area, and production data of the crops are taken into consideration. Five main crops of Bangladesh are considered here as the source of data. The data collection procedure and preprocessing are discussed in detail in the <italic>Methods and measurements</italic> section.</p>
</sec>
<sec id="s6_2">
<title>Model development and prediction</title>
<p>This stage is one of the crucial parts of the crop recommendation system. The preprocessed dataset is used to train the ML models, and the model predicts the production of different crops. The model makes specific crop predictions based on the area, previous production, and environmental data related to the specific crop production period. Our proposed high-performance model KRR described in the <italic>Proposed paradigm</italic> section is considered as the potential ML model in the recommendation system.</p>
<statement id="st2">
<label>Algorithm 2 Crop recommendation</label>
<p>
<preformat>
<bold>&#x2003;Input:</bold> <italic>&#x3b8;</italic> <bold>(Train model);</bold> <italic>&#x3be;</italic> <bold>(Environmental data);</bold> <italic>&#x3b1;</italic> <bold>(Land area);</bold> <italic>&#x3bd;</italic> <bold>(Crops&#x2019; names);</bold> <italic>&#x3c4;</italic>
<bold>(Threshold value)</bold>
<bold>Output: Recommended crop</bold>
1: <bold>procedure</bold> <italic>CropRecommendation (&#x3b8;, &#x3be;, &#x3b1;, &#x3bd;, &#x3c4;)</italic>
2: &#x2003;<bold>for</bold> each <italic>&#x3bd;</italic> <bold>do</bold>
3: &#x2003;&#x2003;<italic>P<sup>&#x3bd;</sup>
</italic> &#x2190; <italic>Predict</italic>(<italic>&#x3b8;,&#x3be;,&#x3b1;</italic>); [<italic>P<sup>&#x3bd;</sup>
</italic> Predicted production amount of <italic>&#x3bd;]</italic>
4: &#x2003;<bold>end for</bold>
5: &#x2003;<bold>if</bold> <italic>P<sup>&#x3bd;</sup>
</italic> &#x2265; <italic>&#x3c4;</italic> <bold>then</bold>
6: &#x2003;&#x2003;return <italic>&#x3bd;</italic>
7: &#x2003;<bold>end if</bold>
8: <bold>end procedure</bold>
</preformat>
</p>
</statement>
</sec>
<sec id="s6_3">
<title>Threshold value selection</title>
<p>We consider the standard environmental data as a threshold for the recommendation. In a specific area, crop production depends on environmental properties like temperature, rainfall, wind speed, and sunshine. The environmental variables can be predicted by our proposed model besides the crop production prediction. Threshold values are taken from the local agriculture office and set to our model for the recommendation.</p>
</sec>
<sec id="s6_4">
<title>Crop recommendation</title>
<p>It is the final step of the recommender system. The model predicts the production for an area, and we set the threshold for this system. The threshold is compared with the predicted environmental data, and the crops&#x2019; production is considered the main element of recommendation. The recommendation is reached by comparing the threshold with the predicted data for each season&#x2019;s crop. The top match is the recommended crop for the season in the area. The input of this system is the season and the list of crops, and the output is the recommended crop among the selected crops. As an extension of this work, in the future, a mobile application will be developed that will help farmers gain easy access to this system. However, in this current state, they can use it with the help of experts with a technical knowledge of putting the inputs and synchronizing with this system.</p>
</sec>
</sec>
<sec id="s7" sec-type="conclusions">
<title>Conclusion</title>
<p>In this work, we have focused on designing a learnable dataset on agricultural crop production prediction from different agricultural research organizations as well as the meteorological department of Bangladesh. The analysis is first performed using five popular classical ML algorithms as well as ensemble ML algorithms. Then, we proposed an ensemble algorithm, called KRR, to better predict crop production. After evaluating all the algorithms, we have found that our proposed ensemble method KRR outperforms the investigated traditional ML and ensemble ML algorithms. In particular, KRR shows minimum errors and a maximum <italic>R</italic>
<sup>2</sup> score compared to that of the investigated ML approaches. We have also provided a DM test to demonstrate the superiority of our proposed KRR approach over the existing ML approaches. The final result also indicates that the production of rice is increasing day by day, and the production of potatoes is also increasing at a significant rate while the production of wheat is decreasing every year. We have also provided a crop recommender system that recommends the most suitable crops to be cultivated on a particular land in the upcoming season.</p>
</sec>
<sec id="s8">
<title>Limitations</title>
<p>This work focused on predicting major crops than the minor crops due to the lack of available data in the study area. Some factors such as soil properties, production cost, and market price, the data collection process of which is difficult and time-consuming, were not considered.</p>
</sec>
<sec id="s9">
<title>Future work</title>
<p>In the future, we will gather more data related to this study and analyze deep learning methodology that can help to select the appropriate crops for the right land more correctly. Local market and wholesale market price analysis will also be performed to select the crops for a specific region. For a more complete picture, researchers plan to include both modern and traditional crops in their future analyses and selections. Additionally, a system based on mobile applications can be created to guarantee that farmers have easy access to the system&#x2019;s information.</p>
</sec>
<sec id="s10" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s11" sec-type="author-contributions">
<title>Author contributions</title>
<p>MH and MM: Conceptualization. MU and MA: Methodology. SK: Software. SM and YN: Validation and formal analysis. All authors contributed to the article and approved the submitted version.</p>
</sec>
</body>
<back>
<sec id="s12" sec-type="funding-information">
<title>Funding</title>
<p>This research was supported by Korea Institute for Advancement of Technology (KIAT) grant funded by the Korea Government (MOTIE) (P0012724, The Competency Development Program for Industry Specialist), the National Research Foundation of Korea (NRF) grant funded by the Korea government (MSIT) (No. RS-2023-00218176), and the Soonchunhyang University Research Fund.</p>
</sec>
<sec id="s13" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s14" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hayat</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ahmad</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kheir</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shaheen</surname> <given-names>F. A.</given-names>
</name>
<name>
<surname>Raza</surname> <given-names>M. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Impact of climate change on dryland agricultural systems: A review of current status, potentials, and further work need</article-title>. <source>Int. J. Plant Production.</source> <volume>p</volume>, <fpage>1</fpage>&#x2013;<lpage>23</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s42106-022-00197-1</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Gaadi</surname> <given-names>K. A.</given-names>
</name>
<name>
<surname>Hassaballa</surname> <given-names>A. A.</given-names>
</name>
<name>
<surname>Tola</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Kayad</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Madugundu</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Alblewi</surname> <given-names>B.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>Prediction of potato crop yield using precision agriculture techniques</article-title>. <source>PloS One</source> <volume>11</volume> (<issue>9</issue>), <fpage>e0162219</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0162219</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bagis</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ustundag</surname> <given-names>B. B.</given-names>
</name>
<name>
<surname>Ozelkan</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>An adaptive spatiotemporal agricultural cropland temperature prediction system based on ground and satellite measurements</article-title>,&#x201d; in <conf-name>2012 First International Conference on Agro-Geoinformatics (AgroGeoinformatics)</conf-name>. <fpage>1</fpage>&#x2013;<lpage>6</lpage> (IEEE). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Agro-Geoinformatics.2012.6311642</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Basak</surname> <given-names>J. K.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>M. N.</given-names>
</name>
<name>
<surname>Rashid</surname> <given-names>M. A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Assessment of the effect of climate change on boro rice production in Bangladesh using DSSAT model</article-title>. <source>J. Civil Eng. (IEB).</source> <volume>38</volume> (<issue>2</issue>), <fpage>95</fpage>&#x2013;<lpage>108</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Benjelloun</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Garcia-Molina</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Kawai</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Larson</surname> <given-names>T. E.</given-names>
</name>
<name>
<surname>Menestrina</surname> <given-names>D.</given-names>
</name>
<etal/>
</person-group>. (<year>2007</year>). &#x201c;<article-title>D-swoosh: A family of algorithms for generic, distributed entity resolution</article-title>,&#x201d; in <conf-name>27th International Conference on Distributed Computing Systems (ICDCS&#x2019;07)</conf-name>. <fpage>37</fpage>&#x2013;<lpage>37</lpage> (IEEE). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICDCS.2007.96</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berger</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Turner</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Siddique</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Knights</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Brinsmead</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Mock</surname> <given-names>I.</given-names>
</name>
<etal/>
</person-group>. (<year>2004</year>). <article-title>Genotype by environment studies across Australia reveal the importance of phenology for chickpea (Cicer arietinum L.) improvement</article-title>. <source>Aust. J. Agric. Res.</source> <volume>55</volume> (<issue>10</issue>), <fpage>1071</fpage>&#x2013;<lpage>1084</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1071/AR04104</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhullar</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Nadeem</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>R. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Simultaneous multi-crop land suitability prediction from remote sensing data using semi-supervised learning</article-title>. <source>Sci. Rep.</source> <volume>13</volume> (<issue>1</issue>), <fpage>6823</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-023-33840-6</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bradter</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Kunin</surname> <given-names>W. E.</given-names>
</name>
<name>
<surname>Altringham</surname> <given-names>J. D.</given-names>
</name>
<name>
<surname>Thom</surname> <given-names>T. J.</given-names>
</name>
<name>
<surname>Benton</surname> <given-names>T. G.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Identifying appropriate spatial scales of predictors in species distribution models with the random forest algorithm</article-title>. <source>Methods Ecol. Evolution.</source> <volume>4</volume> (<issue>2</issue>), <fpage>167</fpage>&#x2013;<lpage>174</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.2041-210x.2012.00253.x</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campbell</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Sands</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ferraro</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Tsao</surname> <given-names>H. Y. J.</given-names>
</name>
<name>
<surname>Mavrommatis</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>From data to action: How marketers can leverage AI</article-title>. <source>Business Horizons.</source> <volume>63</volume> (<issue>2</issue>), <fpage>227</fpage>&#x2013;<lpage>243</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bushor.2019.12.002</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakravarthi</surname> <given-names>B. K.</given-names>
</name>
<name>
<surname>Naravaneni</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>SSR marker based DNA fingerprinting and diversity study in rice (Oryza sativa. L)</article-title>. <source>Afr. J. Biotechnol.</source> <volume>5</volume> (<issue>9</issue>).</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakrobarty</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Al Galib</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>M. Z.</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>M. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Adoption and adaptability of modern aman rice cultivars in faridpur region-Bangladesh</article-title>. <source>Sabrao J. Breed. Genet.</source> <volume>53</volume> (<issue>4</issue>), <fpage>659</fpage>&#x2013;<lpage>672</lpage>. doi: <pub-id pub-id-type="doi">10.54910/sabrao2021.53.4.9</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Charbuty</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Abdulazeez</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Classification based on decision tree algorithm for machine learning</article-title>. <source>J. Appl. Sci. Technol. Trends.</source> <volume>2</volume> (<issue>01</issue>), <fpage>20</fpage>&#x2013;<lpage>28</lpage>. doi: <pub-id pub-id-type="doi">10.38094/jastt20165</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chatrath</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Mishra</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ortiz Ferrara</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Joshi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Challenges to wheat production in South Asia</article-title>. <source>Euphytica.</source> <volume>157</volume> (<issue>3</issue>), <fpage>447</fpage>&#x2013;<lpage>456</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10681-007-9515-2</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cravero</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sepulveda</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Use and adaptations of machine learning in big data&#x2014;Applications in real cases in agriculture</article-title>. <source>Electronics.</source> <volume>10</volume> (<issue>5</issue>), <fpage>552</fpage>. doi: <pub-id pub-id-type="doi">10.3390/electronics10050552</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Danilevicz</surname> <given-names>M. F.</given-names>
</name>
<name>
<surname>Bayer</surname> <given-names>P. E.</given-names>
</name>
<name>
<surname>Nestor</surname> <given-names>B. J.</given-names>
</name>
<name>
<surname>Bennamoun</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Edwards</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Resources for image-based high-throughput phenotyping in crops and data sharing challenges</article-title>. <source>Plant Physiol.</source> <volume>187</volume> (<issue>2</issue>), <fpage>699</fpage>&#x2013;<lpage>715</lpage>. doi: <pub-id pub-id-type="doi">10.1093/plphys/kiab301</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Das</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Saha,</surname> <given-names>S. R.</given-names>
</name>
<name>
<surname>Keya</surname> <given-names>S. S.</given-names>
</name>
<name>
<surname>Suvoni</surname> <given-names>S. S.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Scaling up of jujube-based agroforestry practice and management innovations for improving efficiency and profitability of land uses in Bangladesh</article-title>. <source>Agroforest. Syst.</source> <volume>96</volume> (<issue>2</issue>), <page-range>249&#x2013;263</page-range>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Das</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Management of unanticipated extreme flood: A case study on flooding in NW Bangladesh during 2017</article-title>. <source>Int. J. Disaster Response Emergency Manage. (IJDREM).</source> <volume>1</volume> (<issue>1</issue>), <fpage>22</fpage>&#x2013;<lpage>37</lpage>. doi: <pub-id pub-id-type="doi">10.4018/IJDREM.2018010102</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diebold</surname> <given-names>F. X.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Comparing predictive accuracy, twenty years later: A personal perspective on the use and abuse of Diebold&#x2013;Mariano tests</article-title>. <source>J. Business Economic Statistics.</source> <volume>33</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>1</lpage>. doi: <pub-id pub-id-type="doi">10.1080/07350015.2014.983236</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Faraji</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Amirian Chakan</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Jafarizadeh</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Mohammadian Behbahani</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Soil and nutrient losses due to root crops harvesting: a case study from southwestern Iran</article-title>. <source>Arch. Agron. Soil Science.</source> <volume>63</volume> (<issue>11</issue>), <fpage>1523</fpage>&#x2013;<lpage>1534</lpage>. doi: <pub-id pub-id-type="doi">10.1080/03650340.2017.1296133</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Farooq</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Akbar</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Aslam</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Alyousef</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Predictive modeling for sustainable high-performance concrete from industrial wastes: A comparison and optimization of models using ensemble learners</article-title>. <source>J. Cleaner Production.</source> <volume>292</volume>, <fpage>126032</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jclepro.2021.126032</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Garriga</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Romero-Bravo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Estrada</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Escobar</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Matus</surname> <given-names>I. A.</given-names>
</name>
<name>
<surname>Del Pozo</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Assessing wheat traits by spectral reflectance: do we really need to focus on predicted trait-values or directly identify the elite genotypes group</article-title>? <source>Front. Plant Sci.</source> <volume>8</volume>, <elocation-id>280</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2017.00280</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Glennie</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Lichti</surname> <given-names>D. D.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Static calibration and analysis of the Velodyne HDL-64E S2 for high accuracy mobile scanning</article-title>. <source>Remote sensing.</source> <volume>2</volume> (<issue>6</issue>), <fpage>1610</fpage>&#x2013;<lpage>1624</lpage>. doi: <pub-id pub-id-type="doi">10.3390/rs2061610</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goldstein</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Moses</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sammons</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Birkved</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Potential to curb the environmental burdens of American beef consumption using a novel plant-based beef substitute</article-title>. <source>PloS One</source> <volume>12</volume> (<issue>12</issue>), <elocation-id>e0189029</elocation-id>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0189029</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grange</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Hand</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>1987</year>). <article-title>A review of the effects of atmospheric humidity on the growth of horticultural crops</article-title>. <source>J. Hortic. Science.</source> <volume>62</volume> (<issue>2</issue>), <fpage>125</fpage>&#x2013;<lpage>134</lpage>. doi: <pub-id pub-id-type="doi">10.1080/14620316.1987.11515760</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hancock</surname> <given-names>J. T.</given-names>
</name>
<name>
<surname>Khoshgoftaar</surname> <given-names>T. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>CatBoost for big data: an interdisciplinary review</article-title>. <source>J. big data.</source> <volume>7</volume> (<issue>1</issue>), <fpage>1</fpage>&#x2013;<lpage>45</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s40537-020-00369-8</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hossain</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Abdulla</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Forecasting potato production in Bangladesh by ARIMA model</article-title>. <source>J. Advanced Statistics.</source> <volume>1</volume> (<issue>4</issue>), <fpage>191</fpage>&#x2013;<lpage>198</lpage>. doi: <pub-id pub-id-type="doi">10.22606/jas.2016.14002</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Islam</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Naznin</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Naznin</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Uddin</surname> <given-names>M. N.</given-names>
</name>
<name>
<surname>Amin</surname> <given-names>M. N.</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>M. M.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Dry matter, starch content, reducing sugar, color and crispiness are key parameters of potatoes required for chip processing</article-title>. <source>Horticulturae.</source> <volume>8</volume> (<issue>5</issue>), <fpage>362</fpage>. doi: <pub-id pub-id-type="doi">10.3390/horticulturae8050362</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jansson</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Faiola</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wingler</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X. G.</given-names>
</name>
<name>
<surname>Kravchenko</surname> <given-names>A.</given-names>
</name>
<name>
<surname>De Graaff</surname> <given-names>M. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Crops for carbon farming</article-title>. <source>Front. Plant Science.</source> <volume>12</volume>, <elocation-id>636709</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2021.636709</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jayalakshmi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Gomathi</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Sensor-cloud based precision agriculture approach for intelligent water management</article-title>. <source>Int. J. Plant Production.</source> <volume>14</volume> (<issue>2</issue>), <fpage>177</fpage>&#x2013;<lpage>186</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s42106-019-00077-1</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jha</surname> <given-names>G. K.</given-names>
</name>
<name>
<surname>Sinha</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Agricultural price forecasting using neural network model: An innovative information delivery system</article-title>. <source>Agric. Economics Res. Rev.</source> <volume>26</volume> (<issue>347-2016-17087</issue>), <fpage>229</fpage>&#x2013;<lpage>239</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.22004/ag.econ.162150</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jung</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Maeda</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Bhandari</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ashapure</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Landivar-Bowles</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>The potential of remote sensing and artificial intelligence as tools to improve the resilience of agriculture production systems</article-title>. <source>Curr. Opin. Biotechnol.</source> <volume>70</volume>, <fpage>15</fpage>&#x2013;<lpage>22</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.copbio.2020.09.003</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaur</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Machine learning: applications in Indian agriculture</article-title>. <source>Int. J. Advanced Res. Comput. Communication Engineering.</source> <volume>5</volume> (<issue>4</issue>), <fpage>342</fpage>&#x2013;<lpage>344</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuradusenge</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hitimana</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Hanyurwimfura</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Rukundo</surname> <given-names>P.</given-names>
</name>
<name>
<surname>MTonga</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Mukasine</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Crop yield prediction using machine learning models: case of irish potato and maize</article-title>. <source>Agriculture.</source> <volume>13</volume> (<issue>1</issue>), <fpage>225</fpage>. doi: <pub-id pub-id-type="doi">10.3390/agriculture13010225</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lee</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Moon</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Development of yield prediction system based on real-time agricultural meteorological information</article-title>,&#x201d; in <conf-name>In: 16th international conference on advanced communication technology</conf-name>. <fpage>1292</fpage>&#x2013;<lpage>1295</lpage> (IEEE). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICACT.2014.6779168</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Qian</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Fluctuation characteristics of wheat yield and their relationships with precipitation anoMalies in Anhui province, China</article-title>. <source>Int. J. Plant Production.</source> <volume>16</volume> (<issue>3</issue>), <fpage>483</fpage>&#x2013;<lpage>494</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s42106-022-00203-6</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Minghua</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Qiaolin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhijian</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Jingui</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Prediction model of agricultural product&#x2019;s price based on the improved BP neural network</article-title>,&#x201d; in <conf-name>2012 7th International Conference on Computer Science &amp; Education (ICCSE)</conf-name>. <fpage>613</fpage>&#x2013;<lpage>617</lpage> (IEEE). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCSE.2012.6295150</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Monteiro</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Ramos</surname> <given-names>R. M.</given-names>
</name>
<name>
<surname>Battisti</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Soares</surname> <given-names>J. R.</given-names>
</name>
<name>
<surname>Oliveira</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Figueiredo</surname> <given-names>G. K.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Potential use of data-driven models to estimate and predict soybean yields at national scale in Brazil</article-title>. <source>Int. J. Plant Production.</source> <volume>p</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s42106-022-00209-0</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Morales</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Villalobos</surname> <given-names>F. J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Using machine learning for crop yield prediction in the past or the future</article-title>. <source>Front. Plant Science.</source> <volume>14</volume>, <elocation-id>1128388</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2023.1128388</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nandy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>P. K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Farm efficiency estimation using a hybrid approach of machine-learning and data envelopment analysis: Evidence from rural eastern India</article-title>. <source>J. Cleaner Production.</source> <volume>267</volume>, <fpage>122106</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jclepro.2020.122106</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Panda</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Patra</surname> <given-names>M. R.</given-names>
</name>
</person-group> (<year>2008</year>). &#x201c;<article-title>A comparative study of data mining algorithms for network intrusion detection</article-title>,&#x201d; in <conf-name>2008 First International Conference on Emerging Trends in Engineering and Technology</conf-name>. <fpage>504</fpage>&#x2013;<lpage>507</lpage> (IEEE). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICETET.2008.80</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paudel</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Boogaard</surname> <given-names>H.</given-names>
</name>
<name>
<surname>de Wit</surname> <given-names>A.</given-names>
</name>
<name>
<surname>van der Velde</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Claverie</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Nisini</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Machine learning for regional crop yield forecasting in Europe</article-title>. <source>Field Crops Res.</source> <volume>276</volume>, <fpage>108377</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.fcr.2021.108377</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pereira</surname> <given-names>F. D. O.</given-names>
</name>
<name>
<surname>Teixeira</surname> <given-names>A. P. D. C.</given-names>
</name>
<name>
<surname>de Medeiros</surname> <given-names>F. D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Do essential oils from plants occurring in the Brazilian Caatinga biome present antifungal potential against dermatophytoses? A systematic review</article-title>. <source>Appl. Microbiol. Biotechnol.</source> <volume>105</volume> (<issue>18</issue>), <fpage>6559</fpage>&#x2013;<lpage>6578</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00253-021-11530-5</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prasad</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Patel</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Danodia</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Crop yield prediction in cotton for regional level using random forest approach</article-title>. <source>Spatial Inf. Res.</source> <volume>29</volume> (<issue>2</issue>), <fpage>195</fpage>&#x2013;<lpage>206</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s41324-020-00346-6</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Provost</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Hibert</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Malet</surname> <given-names>J. P.</given-names>
</name>
<name>
<surname>Stumpf</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Doubre</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Automatic classification of endogenous seismic sources within a landslide body using random forest algorithm</article-title>,&#x201d; in <conf-name>EGU General Assembly Conference Abstracts</conf-name>. <fpage>EPSC2016</fpage>&#x2013;<lpage>15705</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ratanamahatana</surname> <given-names>C. A.</given-names>
</name>
<name>
<surname>Gunopulos</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Feature selection for the naive bayesian classifier using decision trees</article-title>. <source>Appl. Artif. Intell.</source> <volume>17</volume> (<issue>5-6</issue>), <fpage>475</fpage>&#x2013;<lpage>487</lpage>. doi: <pub-id pub-id-type="doi">10.1080/713827175</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Razzaghi</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Roderick</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Safro</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Marko</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Multilevel weighted support vector machine for classification on healthcare data with missing values</article-title>. <source>PloS One</source> <volume>11</volume> (<issue>5</issue>), <fpage>e0155119</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0155119</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Royston</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Altman</surname> <given-names>D. G.</given-names>
</name>
<name>
<surname>Sauerbrei</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Dichotomizing continuous predictors in multiple regression: a bad idea</article-title>. <source>Stat Med.</source> <volume>25</volume> (<issue>1</issue>), <fpage>127</fpage>&#x2013;<lpage>141</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/sim.2331</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sagi</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Rokach</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Ensemble learning: A survey</article-title>. <source>Wiley Interdiscip. Reviews: Data Min. Knowledge Discovery.</source> <volume>8</volume> (<issue>4</issue>), <fpage>e1249</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/widm.1249</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sarker</surname> <given-names>M. A. R.</given-names>
</name>
<name>
<surname>Alam</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Gow</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Performance of rain-fed Aman rice yield in Bangladesh in the presence of climate change</article-title>. <source>Renewable Agric. Food systems.</source> <volume>34</volume> (<issue>4</issue>), <fpage>304</fpage>&#x2013;<lpage>312</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1017/S1742170517000473</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Shakoor</surname> <given-names>M. T.</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Rayta</surname> <given-names>S. N.</given-names>
</name>
<name>
<surname>Chakrabarty</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Agricultural production output prediction using supervised machine learning techniques</article-title>,&#x201d; in <conf-name>2017 1st international conference on next generation computing applications (NextComp)</conf-name>. <fpage>182</fpage>&#x2013;<lpage>187</lpage> (IEEE). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/NEXTCOMP.2017.8016196</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Bing</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A hybrid short-term traffic flow prediction model based on singular spectrum analysis and kernel extreme learning machine</article-title>. <source>PloS One</source> <volume>11</volume> (<issue>8</issue>), <elocation-id>e0161259</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0161259</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sharma</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Kamble</surname> <given-names>S. S.</given-names>
</name>
<name>
<surname>Gunasekaran</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A systematic literature review on machine learning applications for sustainable agriculture supply chain performance</article-title>. <source>Comput. Operations Res.</source> <volume>119</volume>, <fpage>104926</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cor.2020.104926</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shehadeh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Alshboul</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Al Mamlook</surname> <given-names>R. E.</given-names>
</name>
<name>
<surname>Hamedat</surname> <given-names>O.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Machine learning models for predicting the residual value of heavy construction equipment: An evaluation of modified decision tree, LightGBM, and XGBoost regression</article-title>. <source>Automation Construction.</source> <volume>129</volume>, <fpage>103827</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.autcon.2021.103827</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Siddique</surname> <given-names>M. N. E. A.</given-names>
</name>
<name>
<surname>de Bruyn</surname> <given-names>L. A. L.</given-names>
</name>
<name>
<surname>Osanai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Guppy</surname> <given-names>C. N.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Typology of rice-based cropping systems for improved soil carbon management: Capturing smallholder farming opportunities and constraints in Dinajpur, Bangladesh</article-title>. <source>Geoderma Regional.</source> <volume>28</volume>, <elocation-id>e00460</elocation-id>. doi: <pub-id pub-id-type="doi">10.1016/j.geodrs.2021.e00460</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Somvanshi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Chavan</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Tambade</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Shinde</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>A review of machine learning techniques using decision tree and support vector machine</article-title>,&#x201d; in <conf-name>2016 international conference on computing communication control and automation (ICCUBEA)</conf-name>. <fpage>1</fpage>&#x2013;<lpage>7</lpage> (IEEE). doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCUBEA.2016.7860040</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stekhoven</surname> <given-names>D. J.</given-names>
</name>
<name>
<surname>Buhlmann</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>MissForest&#x2014;non-parametric missing value imputation for mixed-type data</article-title>. <source>Bioinformatics.</source> <volume>28</volume> (<issue>1</issue>), <fpage>112</fpage>&#x2013;<lpage>118</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btr597</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sujjaviriyasup</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Pitiruek</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Agricultural product forecasting using machine learning approach</article-title>. <source>Int. J. Math Analysis.</source> <volume>7</volume> (<issue>38</issue>), <fpage>1869</fpage>&#x2013;<lpage>1875</lpage>. doi: <pub-id pub-id-type="doi">10.12988/ijma.2013.35113</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tavares</surname> <given-names>O. C. H.</given-names>
</name>
<name>
<surname>Santos</surname> <given-names>L. A.</given-names>
</name>
<name>
<surname>Filho</surname> <given-names>D. F.</given-names>
</name>
<name>
<surname>Ferreira</surname> <given-names>L. M.</given-names>
</name>
<name>
<surname>Garcia</surname> <given-names>A. C.</given-names>
</name>
<name>
<surname>Castro</surname> <given-names>T. A. V. T.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Response surface modeling of humic acid stimulation of the rice (Oryza sativa L.) root system</article-title>. <source>Arch. Agron. Soil Sci.</source> <volume>67</volume> (<issue>8</issue>), <fpage>1046</fpage>&#x2013;<lpage>1059</lpage>. doi: <pub-id pub-id-type="doi">10.1080/03650340.2020.1775199</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uddin</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Matin</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Meyer</surname> <given-names>F. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Operational flood mapping using multitemporal Sentinel-1 SAR images: A case study from Bangladesh</article-title>. <source>Remote Sensing.</source> <volume>11</volume> (<issue>13</issue>), <fpage>1581</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs11131581</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Ittersum</surname> <given-names>M. K.</given-names>
</name>
<name>
<surname>Cassman</surname> <given-names>K. G.</given-names>
</name>
<name>
<surname>Grassini</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Wolf</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Tittonell</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Hochman</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Yield gap analysis with local to global relevance&#x2014;a review</article-title>. <source>Field Crops Res.</source> <volume>143</volume>, <fpage>4</fpage>&#x2013;<lpage>17</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.fcr.2012.09.009</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Klompenburg</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Kassahun</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Catal</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Crop yield prediction using machine learning: A systematic literature review</article-title>. <source>Comput. Electron. Agriculture.</source> <volume>177</volume>, <fpage>105709</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105709</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Prediction of winter wheat yield and dry matter in North China Plain using machine learning algorithms for optimal water and nitrogen application</article-title>. <source>Agric. Water Management.</source> <volume>277</volume>, <fpage>108140</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.agwat.2023.108140</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Young</surname> <given-names>L. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Agricultural crop forecasting for large geographical areas</article-title>. <source>Annu. Rev. Stat its application.</source> <volume>6</volume>, <fpage>173</fpage>&#x2013;<lpage>196</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev-statistics-030718-105002</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ai</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Croft</surname> <given-names>W. B.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Towards conversational search and recommendation: System ask, user respond</article-title>,&#x201d; in <conf-name>Proceedings of the 27th acm international conference on information and knowledge management</conf-name>. <fpage>177</fpage>&#x2013;<lpage>186</lpage>.</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>California almond yield prediction at the orchard level with a machine learning approach</article-title>. <source>Front. Plant science.</source> <volume>10</volume>, <elocation-id>809</elocation-id>. doi: <pub-id pub-id-type="doi">10.3389/fpls.2019.00809</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>