<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Environ. Sci.</journal-id>
<journal-title>Frontiers in Environmental Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Environ. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-665X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1213069</article-id>
<article-id pub-id-type="doi">10.3389/fenvs.2023.1213069</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Environmental Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Improving prediction accuracy for acid sulfate soil mapping by means of variable selection</article-title>
<alt-title alt-title-type="left-running-head">Est&#xe9;vez et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenvs.2023.1213069">10.3389/fenvs.2023.1213069</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Est&#xe9;vez</surname>
<given-names>Virginia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2062530/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mattb&#xe4;ck</surname>
<given-names>Stefan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Boman</surname>
<given-names>Anton</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Beucher</surname>
<given-names>Am&#xe9;lie</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1251745/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bj&#xf6;rk</surname>
<given-names>Kaj-Mikael</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1786872/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>&#xd6;sterholm</surname>
<given-names>Peter</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Graduate School and Research, Arcada University of Applied Sciences</institution>, <addr-line>Helsinki</addr-line>, <country>Finland</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Geology and Mineralogy</institution>, <institution>&#xc5;bo Akademi University</institution>, <addr-line>&#xc5;bo</addr-line>, <country>Finland</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Geological Survey of Finland</institution>, <addr-line>Kokkola</addr-line>, <country>Finland</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Agroecology</institution>, <institution>Aarhus University</institution>, <addr-line>Tjele</addr-line>, <country>Denmark</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/81071/overview">Alexander Kokhanovsky</ext-link>, German Research Centre for Geosciences, Germany</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1255538/overview">Calogero Schillaci</ext-link>, Joint Research Centre, Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1413508/overview">Shamsollah Ayoubi</ext-link>, Isfahan University of Technology, Iran</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Virginia Est&#xe9;vez, <email>estevezv@arcada.fi</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>14</day>
<month>07</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1213069</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>04</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>04</day>
<month>07</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Est&#xe9;vez, Mattb&#xe4;ck, Boman, Beucher, Bj&#xf6;rk and &#xd6;sterholm.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Est&#xe9;vez, Mattb&#xe4;ck, Boman, Beucher, Bj&#xf6;rk and &#xd6;sterholm</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Acid sulfate soils can cause environmental damage and geotechnical problems when drained or exposed to oxidizing conditions. This makes them one of the most harmful soils found in nature. In order to reduce possible damage derived from this type of soil, it is fundamental to create occurrence maps showing their localization. Nowadays, occurrence maps can be created using machine learning techniques. The accuracy of these maps depends on two factors: the dataset and the machine learning method. Previously, different machine learning methods were evaluated for acid sulfate soil mapping. To improve the precision of the acid sulfate soil probability maps, in this qualitative modeling study we have added more environmental covariates (17 in total). Since a greater number of covariates does not necessarily imply an improvement in the prediction, we have selected the most relevant environmental covariates for the classification and prediction of acid sulfate soils. For this, we have applied eleven different variable selection methods. The predictive abilities of each group of selected variables have been analyzed using Random Forest and Gradient Boosting. We show that the selection of each environmental covariate as well as the relationship between them are extremely important for an accurate prediction of acid sulfate soils. Among the variable selection methods analyzed, Random Forest stands out, as it is the one that has best selected the relevant covariates for the classification of these soils. Furthermore, the combination of two variable selection methods can improve the prediction of the model. Contrary to the general belief, a low correlation between the covariates does not guarantee a good performance of the model. In general, Random Forest has given better results in the prediction than Gradient Boosting. From the best results obtained, an acid sulfate soils occurrence map has been created. Compared with previous studies in the same area, variable selection has improved the accuracy by 15%&#x2013;17% for the models based on Random Forest. The present study confirms the importance of variable selection for the prediction of acid sulfate soils.</p>
</abstract>
<kwd-group>
<kwd>variable selection</kwd>
<kwd>acid sulfate soils</kwd>
<kwd>machine learning</kwd>
<kwd>digital soil mapping</kwd>
<kwd>random forest</kwd>
<kwd>gradient boosting</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Environmental Informatics and Remote Sensing</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>In general terms, soils that present sulfidic materials in their composition and a drop or possible drop in their pH values below 4 are considered acid sulfate (AS) soils (<xref ref-type="bibr" rid="B69">Pons, 1973</xref>). The drop in pH values is a consequence of the oxidation of sulfidic materials. The oxidation process is often initiated by drainage of the soils in agriculture or forestry. A decrease of the soil-pH below 4 generates acidification of the soil and mobilization of several metals, and ultimately leaching of acidity and metals from the soil through subsurface drainpipes and ditches (<xref ref-type="bibr" rid="B5">&#xc5;str&#xf6;m and Bj&#xf6;rklund, 1997</xref>; <xref ref-type="bibr" rid="B64">&#xd6;sterholm and &#xc5;str&#xf6;m, 2002</xref>; <xref ref-type="bibr" rid="B73">Roos and &#xc5;str&#xf6;m, 2006</xref>). Therefore, the occurrence of AS soils can lead to the deterioration of stream waters, which may cause severe ecological damages (e.g., fish kills (<xref ref-type="bibr" rid="B44">Hudd, 2000</xref>; <xref ref-type="bibr" rid="B80">Urho, 2002</xref>)), as well as problems in agriculture and its productivity (<xref ref-type="bibr" rid="B66">Palko, 1994</xref>) or in infrastructures with damages related to the poor stability of sulfidic sediments and corrosion of concrete and steel constructions as a consequence of the increased acidity. Due to these environmental hazards and geotechnical challenges, AS soils are considered one of the most damaging soils (<xref ref-type="bibr" rid="B60">Michael, 2013</xref>). For environmental authorities, mapping of AS soil occurrence is very important to identify areas with potential environmental hazards such as acidifying and metal pollution if the soil materials are disturbed. The identification (mapping) of these areas would contribute to the reduction of possible ecological damages. In infrastructure developments, the knowledge of AS soil occurrence is crucial to determine the need or not to apply measures to avoid issues related to the poor stability and corrosion of building materials, which often lead to increased building costs.</p>
<p>Traditional or conventional methods for mapping AS soils require a large number of soil samples as well as expert knowledge to create the maps. As a result, the mapping process can be very laborious, expensive and time consuming. Moreover, the accuracy of the resulting maps can be affected by the person who creates the map. Nowadays, digital soil mapping studies are mostly based on the use of machine learning techniques (<xref ref-type="bibr" rid="B59">McBratney et al., 2003</xref>). These techniques have several advantages over traditional methods. First, mapping with these methods requires fewer samples (<xref ref-type="bibr" rid="B18">Brus et al., 2011</xref>), leading to a much faster and less expensive process. Moreover, these techniques make the mapping process more objective and easier to replicate than traditional methods. So far, the most used machine learning techniques for AS soil mapping have been Artificial Neural Network (ANN) (<xref ref-type="bibr" rid="B13">Beucher et al., 2013</xref>; <xref ref-type="bibr" rid="B14">Beucher et al., 2015</xref>; <xref ref-type="bibr" rid="B11">Beucher et al., 2017</xref>), Fuzzy logic (<xref ref-type="bibr" rid="B12">Beucher et al., 2014</xref>) and Fuzzy k-means (<xref ref-type="bibr" rid="B43">Huang et al., 2014</xref>). Although there are few works, other techniques such as Convolutional Neural Network (CNN) (<xref ref-type="bibr" rid="B25">Est&#x00E9;vez Nu&#x00F1;o, 2020</xref>; <xref ref-type="bibr" rid="B10">Beucher et al., 2022</xref>) or Extreme Learning Machine (ELM) (<xref ref-type="bibr" rid="B27">Est&#xe9;vez et al., 2023</xref>; <xref ref-type="bibr" rid="B3">Akusok et al., 2023</xref>) have also been studied for the classification of AS soils. Recently, the suitability of three different machine learning methods, Random Forest (RF), Gradient Boosting (GB) and Support Vector Machines (SVM), for the classification and prediction of AS soils has been analyzed (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). RF and GB showed high abilities for the classification and prediction of AS soils, leading to very accurate AS soil probability maps. On the contrary, it has been shown that SVM is unsuitable for mapping AS soils due to the fact that it could not adequately recognize AS soils (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). One of the main goals of this study is to enhance the accuracy of the AS soil probability maps. For this reason, the study has been focused on improving RF and GB models using more environmental covariates. In principle, the consideration of a greater number of covariates will give more information for a better characterization of the soils. However, it should be noted that some covariates may be irrelevant or provide redundant information that could lead to a poor prediction (<xref ref-type="bibr" rid="B39">Hall and Holmes, 2003</xref>). Thus, the objective is to find the most relevant environmental covariates that allow the classification and prediction of the AS soils with a very high accuracy. For this, it will be essential to apply variable selection, which is one of the most important and complex topics in machine learning. Variable selection is not only fundamental for improving the prediction of the model but also to understand the relationship between the variables and the target, as well as to reduce the computing time (<xref ref-type="bibr" rid="B36">Guyon and Elisseeff, 2003</xref>). Variable selection has been widely used in different fields such as biomedicine and bioinformatics (<xref ref-type="bibr" rid="B37">Guyon et al., 2002</xref>; <xref ref-type="bibr" rid="B74">Saeys et al., 2007</xref>; <xref ref-type="bibr" rid="B63">Osl et al., 2009</xref>) or text classification problems (<xref ref-type="bibr" rid="B29">Forman, 2003</xref>). In soil science, the variable selection have been used for the prediction of soil organic carbon (<xref ref-type="bibr" rid="B86">Xiong et al., 2014</xref>; <xref ref-type="bibr" rid="B28">Fitzpatrick et al., 2016</xref>; <xref ref-type="bibr" rid="B55">Lie et al., 2016</xref>; <xref ref-type="bibr" rid="B46">Keskin et al., 2019</xref>), soil parent material (<xref ref-type="bibr" rid="B41">Heung et al., 2014</xref>), soil organic matter (<xref ref-type="bibr" rid="B23">Chen et al., 2022</xref>), soil depth (<xref ref-type="bibr" rid="B78">Tesfa et al., 2009</xref>; <xref ref-type="bibr" rid="B19">Camera et al., 2017</xref>; <xref ref-type="bibr" rid="B22">Castro Franco et al., 2017</xref>; <xref ref-type="bibr" rid="B56">Lu et al., 2019</xref>) and soil classes (<xref ref-type="bibr" rid="B9">Behrens et al., 2010</xref>; <xref ref-type="bibr" rid="B17">Brungard et al., 2015</xref>; <xref ref-type="bibr" rid="B19">Camera et al., 2017</xref>; <xref ref-type="bibr" rid="B21">Campos et al., 2018</xref>). So far, there are hardly any works where the selection of variables has been applied for the prediction of AS soils. In this study, the variable selection for an accurate prediction of AS soils has been analyzed in detail. As an environmental covariate can be relevant for one method but irrelevant for another (<xref ref-type="bibr" rid="B47">Kohavi and John, 1997</xref>), eleven different methods of variable selection have been considered. This allows the identification of the most relevant covariates for the characterization of the AS soils and their prediction. The methods used are: one Univariate Feature Selection (UFS), RF, GB, Extra Trees Classifier (ETC), Recursive Feature Elimination (RFE), Backward Selection and Pearson&#x2019;s correlation. For RFE, four different methods have been considered: RF, GB, ETC and Logistic Regression (LR). The Backward Selection has been analyzed for two different methods, RF and GB. Moreover, the combination of two variable selection methods has also been studied. Once the subsets of environmental covariates were selected, their suitability for the prediction and classification of AS soils for the two machine learning methods considered for modeling, RF and GB, has been evaluated. The AS soil probability map for the study area has been created using the model and the group of environmental covariates with the highest abilities for the classification of AS soils. Finally, the extent of AS soils has been estimated based on the modeled AS soil probability map.</p>
</sec>
<sec sec-type="materials|methods" id="s2">
<title>2 Materials and methods</title>
<sec id="s2-1">
<title>2.1 Study area</title>
<p>The area considered for this study is Virolahti and its surroundings (1,091&#xa0;km<sup>2</sup>), located in southeastern Finland (<xref ref-type="fig" rid="F1">Figure 1</xref>). The land use of this region, which is part of the boreal ecosystem, is mainly forestry, agricultural lands and some urban areas. In this area, as in the rest of Finland, AS soils belong to the cryic soil temperature regime, where mean annual soil temperature 0&#xb0;C&#x2013;8&#xb0;C and mean summer soil temperature below 15&#xb0;C (<xref ref-type="bibr" rid="B87">Yli-Halla and Mokma, 1998</xref>). About 905&#xa0;km<sup>2</sup> (83%) of the study area corresponds to the Littorina Sea maximum extent, which is the most potential area for AS soil occurrence and where the conventional AS soil mapping has been made by the Geological Survey of Finland (GTK). The geological basement is composed almost entirely of 1.66&#x2013;1.60&#xa0;Ga Rapakivi granite (<xref ref-type="bibr" rid="B51">Lehtinen et al., 1998</xref>; <xref ref-type="bibr" rid="B32">Geological Survey of Finland, 2021</xref>) and is covered mainly by glacial till and alluvial deposits (<xref ref-type="bibr" rid="B38">Haavisto-Hyv&#xe4;rinen and Kutvonen, 2007</xref>). The uppermost meter in the area is made up of bedrock, outcrops and block fields (57.66%), different types of soils (38.94%), water (3.19%) and a small unmapped part (0.22%). The existing soils in this area are clay (16.91%), fine sand to gravel (7.21%), till (5.85%), thick peat deposits (4.63%), gyttja (2.01%), fine-grained sediment or fine silt low humus content 2%&#x2013;6% (1.21%), fine silt (1.12%), and man-made soils.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>(Color online) Main figure: Location of the study area (red color) and extent of the Littorina Sea (diagonal lines). Inset: Location of the training and validation points for acid sulfate (AS) and non acid sulfate (non-AS) soils in the study area.</p>
</caption>
<graphic xlink:href="fenvs-11-1213069-g001.tif"/>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Soil samples</title>
<p>The soil cores used in this study were collected with a gouge auger down to 2&#x2013;3&#xa0;m depth as part of the national AS soil mapping done by GTK. The emplacement of the sampling sites was limited to the Littorina Sea maximal extent, as the majority of the Finnish AS soils are located here (<xref ref-type="bibr" rid="B66">Palko, 1994</xref>; <xref ref-type="bibr" rid="B88">Yli-Halla et al., 1999</xref>; <xref ref-type="bibr" rid="B32">Geological Survey of Finland, 2021</xref>). The non-statistical sampling plan used was designed to create a set of samples where the different types of soils and materials of the area were included. This was possible thanks to the consideration of all sediment classes in the quaternary geology map, different topography locations, and the electric conductivity (EC) anomalies and non-anomalies in airborne geophysical data. The sampling density during the AS soil mapping in Finland is about 1 probe/km<sup>2</sup>. The exclusion of bedrock, outcrops, glacial till, man-made soils and water from the sampling, as well as the limited road network, lead to areas where the sampling density is less dense.</p>
<p>The classification of the soil samples or cores into AS and non-AS soils was determined by the soil-pH, which was measured in the field and/or in the laboratory after oxidation (incubation). Mineral soil materials with pH <inline-formula id="inf1">
<mml:math id="m1">
<mml:mo>&#x3c;</mml:mo>
</mml:math>
</inline-formula> 4 and organic soil materials with pH <inline-formula id="inf2">
<mml:math id="m2">
<mml:mo>&#x3c;</mml:mo>
</mml:math>
</inline-formula> 3 were classified as AS soils, whereas samples having pH-values above these were classified as non-AS soils, following slightly modified procedures described in (<xref ref-type="bibr" rid="B15">Boman et al., 2019</xref>).</p>
</sec>
<sec id="s2-3">
<title>2.3 Environmental covariates</title>
<p>In addition to soil samples, environmental data have been used in this study. The environmental covariates are raster data generated from remote sensing data. In this study, the types of remote sensing data used are LiDAR and geophysics, which originate from airborne surveys. Some environmental covariates can be fundamental to characterize different types of soils. These covariates are of several types: Quaternary geology, airborne electromagnetic or aerogeophysics data, digital elevation model and topographic or terrain data. In this study, a total of 17 environmental layers have been used (<xref ref-type="table" rid="T2">Table 2</xref>). Contrary to a previous work where only one terrain layer was considered (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>), twelve different terrain layers have been analyzed for the classification of AS soils in this study. All environmental covariates have a resolution of 50&#xa0;m &#xd7; 50&#xa0;m and have been created using Qgis (<xref ref-type="bibr" rid="B72">Qgis Development Team, 2019</xref>). Furthermore, the coordinate reference system is the Finnish one (ETRS89/TM35FIN(E,N)).</p>
<sec id="s2-3-1">
<title>2.3.1 Quaternary</title>
<p>The quaternary geology layer 1:200000 (<xref ref-type="bibr" rid="B48">Korpela and Niemel&#xe4;, 1985</xref>) displays the occurrence of 12 different soil materials down to a depth of 1&#xa0;m. Fine-grained gyttja bearing sediment is the most critical indicator of AS soils because they usually consist of fine-grained sediments, although in some environments it may be made up of coarse-grained soil materials (<xref ref-type="bibr" rid="B58">Mattb&#xe4;ck et al., 2017</xref>).</p>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Aerogeophysics</title>
<p>Airborne electromagnetic data (real, imaginary and apparent resistivity components) are often useful for discovering sulfidic deposits, both in the overburden and bedrock (<xref ref-type="bibr" rid="B1">Airo, 2005</xref>). The aeroelectromagnetic data were collected at flight altitudes between 30 and 40&#xa0;m and a line spacing of 200&#xa0;m, producing a raster dataset with a resolution of 50&#xa0;m. Shallow weak anomalies, that mainly relate to variations in topsoil thickness and electric conductivity, may be detected using the imaginary component whereas the real component mainly enables the detection of deeper anomalies in the bedrock, e.g., black schists (<xref ref-type="bibr" rid="B2">Airo and Loukola-Ruskeeniemi, 2004</xref>). The apparent resistivity is calculated from In-Phase (real) and Quadrature (imaginary) components of the measured electromagnetic field. Soil materials such as clay and gyttja have high conductivities whereas glacial till and sand have lower (<xref ref-type="bibr" rid="B68">Pernu, 1991</xref>).</p>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Digital elevation model</title>
<p>A digital elevation model (DEM) is a representation of the topographic surface of the terrain. This environmental covariate will play a fundamental role in the classification of AS soils since in southern Finland, AS soils generally occurs at an elevation of less than 50&#xa0;m (<xref ref-type="bibr" rid="B66">Palko, 1994</xref>). The DEM used in this study has been generated from the LiDAR data of the National Land Survey of Finland (NLS). The resolution of this layer is 2&#xa0;m &#xd7; 2&#xa0;m but was down-sampled to a 50&#xa0;m &#xd7; 50&#xa0;m which is the resolution of the aerogeophysics layers and therefore the one used in this study. The resolution change has been done in Qgis by reprojection and the resampling method used is Nearest Neighbors.</p>
</sec>
<sec id="s2-3-4">
<title>2.3.4 Terrain layers</title>
<p>The terrain attributes are obtained from the DEM, and widely used for classification and prediction in digital soil mapping (<xref ref-type="bibr" rid="B59">McBratney et al., 2003</xref>). In this study 12 terrain layers have been considered: slope, aspect, hillshade, roughness, multiresolution index of valley bottom flatness (MRVBF), multiresolution index of the ridge top flatness (MRRTF), topographic position index (TPI), terrain ruggedness index (TRI), topographic wetness index (TWI), valley depth, tangential curvature and profile curvature. In Finland, AS soils usually occur in low-relief, flat and heterogenous areas, usually with a close to zero-degree slope (<xref ref-type="bibr" rid="B65">&#xd6;sterholm et al., 2005</xref>; <xref ref-type="bibr" rid="B8">Becher et al., 2018</xref>). Covariates such as hillshade, slope, roughness and terrain ruggedness show this. However, it is not clear if aspect, profile curvature or tangential curvature will impact AS soil occurrence, since the direction and type of ridge do not really affect the occurrence. On the other hand, valley bottom flatness and valley depth covariates are important since they reflect depositional environments where AS soils most likely have formed. Contrary, ridge top flatness might not have that big of an impact since sulfidic sediments usually do not form on ridge tops in Finland. While topographic position index should not have that much of an impact, the wetness index is somewhat similar to valley bottom flatness and valley depth, which usually indicate waterlogged soils where AS soils are often found. Exceptions may occur in sandy areas where coarse-grained AS soils consist of littoral deposits and, where beach ridges and dunes are common (<xref ref-type="bibr" rid="B58">Mattb&#xe4;ck et al., 2017</xref>).</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Machine learning methods for modeling</title>
<p>For the modeling, two machine learning techniques, Random Forest (RF) and Gradient Boosting (GB), have been considered. These two ensemble methods based on decision trees have shown high performance abilities for the prediction of AS soils (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). In this work, Python (<xref ref-type="bibr" rid="B81">Van Rossum and Drake, 2009</xref>) has been used for all the codes and the Scikit-learn library (<xref ref-type="bibr" rid="B67">Pedregosa et al., 2011</xref>) for the machine learning methods. For the optimal performance of machine learning models, tuning parameters are critical (<xref ref-type="bibr" rid="B62">M&#xfc;ller and Guido, 2016</xref>). The determination of the best tuning parameters for the two machine learning models have been made with grid search and cross-validation (GridSearchCV). For more information we refer the reader to see the previous work (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>).</p>
<sec id="s2-4-1">
<title>2.4.1 Random forest</title>
<p>Random Forest (RF) (<xref ref-type="bibr" rid="B16">Breiman, 2001</xref>) is one of the most used techniques in classification and regression problems due to its efficiency and robustness. This supervised machine learning technique combines the results of multiple decision trees to make a prediction. Each tree is created from a different sub-dataset, which has been randomly selected. For each tree, the algorithm will make a prediction that will be considered for the final prediction. This leads to a better performance than in the case of a single tree. Moreover, this technique helps to reduce overfitting. In soil science, this method has been frequently used for the prediction of soil properties (<xref ref-type="bibr" rid="B35">Grimm et al., 2008</xref>; <xref ref-type="bibr" rid="B9">Behrens et al., 2010</xref>; <xref ref-type="bibr" rid="B85">Wiesmeier et al., 2011</xref>; <xref ref-type="bibr" rid="B54">Lie et al., 2012</xref>; <xref ref-type="bibr" rid="B75">Schmidt et al., 2014</xref>; <xref ref-type="bibr" rid="B82">Veronesi and Schillaci, 2019</xref>; <xref ref-type="bibr" rid="B6">Azizi et al., 2022</xref>; <xref ref-type="bibr" rid="B61">Moradpour et al., 2023</xref>) and the classification of soils (<xref ref-type="bibr" rid="B41">Heung et al., 2014</xref>; <xref ref-type="bibr" rid="B17">Brungard et al., 2015</xref>; <xref ref-type="bibr" rid="B31">Gambill et al., 2016</xref>; <xref ref-type="bibr" rid="B42">Heung et al., 2016</xref>; <xref ref-type="bibr" rid="B77">Teng et al., 2018</xref>; <xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>).</p>
</sec>
<sec id="s2-4-2">
<title>2.4.2 Gradient boosting</title>
<p>For classification and regression predictions, one common supervised machine learning technique is Gradient Boosting (GB)(<xref ref-type="bibr" rid="B30">Friedman, 2001</xref>). This method is very efficient and can lead to predictions with very high precision if the tuning parameters are adequate. Unlike RF, GB builds the ensemble trees one by one, based on the information of the previous tree. This serial manner allows each new tree to correct the prediction errors made in the previous one. The goal of the model is to improve the final prediction. In soil science, this method has been considered to predict soil properties (<xref ref-type="bibr" rid="B40">Hengl et al., 2017</xref>; <xref ref-type="bibr" rid="B76">Sindayiheburaa et al., 2017</xref>; <xref ref-type="bibr" rid="B79">Tziachrisa et al., 2019</xref>) and classes (<xref ref-type="bibr" rid="B52">Lemercier et al., 2012</xref>; <xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>).</p>
</sec>
</sec>
<sec id="s2-5">
<title>2.5 Variable selection methods</title>
<p>The selection of the most important covariates for the study is essential for a good classification of soils. There are covariates that do not give information to the model and can hinder the prediction (<xref ref-type="bibr" rid="B39">Hall and Holmes, 2003</xref>). Thus, the variable selection is fundamental for the creation of an effective predictive model. The set of the most relevant environmental covariates for the model depends on the variable selection method. Moreover, an optimal covariate set can be more appropriate for a given machine learning method than for another. In the case of classification studies, the variable selection methods are of three types: filter models, wrapper and embedded methods (<xref ref-type="bibr" rid="B74">Saeys et al., 2007</xref>; <xref ref-type="bibr" rid="B49">Kuhn and Johnson, 2013</xref>). In this study, filter models, wrapper methods and the analysis of the correlation have been considered.</p>
<sec id="s2-5-1">
<title>2.5.1 Filter methods</title>
<p>The variable selection in the filter methods is independent of the machine learning techniques. This independence makes these methods computationally efficient and generally does not lead to overfitting (<xref ref-type="bibr" rid="B74">Saeys et al., 2007</xref>; <xref ref-type="bibr" rid="B49">Kuhn and Johnson, 2013</xref>). However, the set of variables selected cannot be the most suitable for a given model. Another problem is that these methods do not take into account the correlation between the variables. As a result, highly correlated variables can be selected, adding redundant information to the model.</p>
<sec id="s2-5-1-1">
<title>2.5.1.1 Univariate feature selection</title>
<p>These methods select the most relevant features by statistical tests. Univariate refers to the fact that each variable is analyzed individually, without taking into account the relationship between the rest of the variables. Thus, the model measures the relationship of each feature with the target. The resulting values allow the selection of the most relevant variables for the target. There are several methods of univariate selection, in this work the method used is the SelectKBest from Scikit-learn (<xref ref-type="bibr" rid="B67">Pedregosa et al., 2011</xref>).</p>
</sec>
</sec>
<sec id="s2-5-2">
<title>2.5.2 Wrapper methods</title>
<p>The wrapper methods use machine learning techniques for the selection of variables, which is based on the performance of the model. The most appropriate features for the model are those that improve the accuracy. In general, this makes them the best performing variable selection methods. Unlike the filter methods, the wrapper methods can cause an excessive adjustment of the results, i.e., overfitting (<xref ref-type="bibr" rid="B47">Kohavi and John, 1997</xref>). Furthermore, these methods require more computation time. The machine learning method used for the variable selection can be the same or different to the one used for the final modeling. In this study, we have evaluated several wrapper methods for feature selection.</p>
<sec id="s2-5-2-1">
<title>2.5.2.1 Methods based on decision trees</title>
<p>These methods can directly give the importance of each feature for the model by means of an implemented algorithm. In this work, RF and GB have also been used for variable selection. One of the most common methods for variable selection in soil science is RF, that has been used for the prediction of soil organic carbon (<xref ref-type="bibr" rid="B46">Keskin et al., 2019</xref>), soil organic material (<xref ref-type="bibr" rid="B41">Heung et al., 2014</xref>), soil organic matter (<xref ref-type="bibr" rid="B23">Chen et al., 2022</xref>), soil depth (<xref ref-type="bibr" rid="B78">Tesfa et al., 2009</xref>; <xref ref-type="bibr" rid="B22">Castro Franco et al., 2017</xref>; <xref ref-type="bibr" rid="B56">Lu et al., 2019</xref>) or soil thickness (<xref ref-type="bibr" rid="B53">Li et al., 2020</xref>). In contrast, GB has only been used in variable selection for the prediction of soil depth (<xref ref-type="bibr" rid="B50">Lacoste et al., 2016</xref>).</p>
<p>In addition, another method based on decision trees has been considered for variable selection: Extra Trees Classifier (ETC).</p>
<sec id="s2-5-2-1-1">
<title>2.5.2.1.1 Extra trees classifier</title>
<p>Extra trees classifier (ETC) (<xref ref-type="bibr" rid="B33">Geurts et al., 2006</xref>) or extremely randomized trees is an ensemble machine learning technique quite similar to RF. One difference is that in the, ETC, the decision trees are created on the whole data and not on the randomly selected data as in RF. However, the main difference in both methods is the selection of the split points in a decision tree. RF selects an optimal split point taking into account the best features, while, ETC chooses randomly the split points independently of the features. So far, ETC has never been used for classification or prediction of soil classes or properties. In this study, the method is only used for variable selection.</p>
</sec>
</sec>
<sec id="s2-5-2-2">
<title>2.5.2.2 Backward selection</title>
<p>This method is based on the elimination of the irrelevant features. Initially all variables or features are considered, and in each step the least important feature for the model is eliminated. The idea is to improve the accuracy of the model. The elimination of variables will take place until the performance of the model does not improve. In this study, the machine learning techniques used for the backward selection are RF and GB. Previously, Backward Selection has been applied for the prediction of soil organic carbon (<xref ref-type="bibr" rid="B55">Lie et al., 2016</xref>; <xref ref-type="bibr" rid="B82">Veronesi and Schillaci, 2019</xref>).</p>
</sec>
<sec id="s2-5-2-3">
<title>2.5.2.3 Recursive feature elimination</title>
<p>This method is an iterative feature selection based on the elimination of the least important features (<xref ref-type="bibr" rid="B37">Guyon et al., 2002</xref>). This is also a backward selection, but in this case, a subset of the features with higher weight in the model is selected at once. In this way, the variable selection method is optimized due to the features selected are relevant when they are combined together. It should be noted that a feature can be relevant in presence of other features but not by itself (<xref ref-type="bibr" rid="B36">Guyon and Elisseeff, 2003</xref>). Recursive Feature Elimination (RFE) needs other methods to measure the weight of the features. In this study, RFE have been analyzed using four different machine learning techniques: RF, GB, ETC and Logistic Regression (LR). So far, of these methods only the RFE with RF has been used in the variable selection of environmental covariates in soil science (<xref ref-type="bibr" rid="B17">Brungard et al., 2015</xref>; <xref ref-type="bibr" rid="B19">Camera et al., 2017</xref>; <xref ref-type="bibr" rid="B82">Veronesi and Schillaci, 2019</xref>; <xref ref-type="bibr" rid="B10">Beucher et al., 2022</xref>).</p>
<sec id="s2-5-2-3-1">
<title>2.5.2.3.1 Logistic regression</title>
<p>This machine learning method is widely used in binary classification problems (<xref ref-type="bibr" rid="B62">M&#xfc;ller and Guido, 2016</xref>). LR is a linear model that has been used in soil science to predict the occurrence of soil types (<xref ref-type="bibr" rid="B34">Giasson et al., 2006</xref>; <xref ref-type="bibr" rid="B24">Debella-Gilo and Etzelm&#xfc;ller, 2009</xref>), soil drainage classes (<xref ref-type="bibr" rid="B20">Campling et al., 2002</xref>) or diagnostic horizons (<xref ref-type="bibr" rid="B45">Jafari et al., 2012</xref>). However, LR has never been used for variable selection in soil science.</p>
</sec>
</sec>
</sec>
<sec id="s2-5-3">
<title>2.5.3 Correlation</title>
<p>Unlike previous variable selection methods, the correlation is an unsupervised method, which does not take into account the target or label. The correlation indicates the relationship between two variables, which gives information about the redundancy of the variables. In the case of highly correlated variables the information they provide is redundant (<xref ref-type="bibr" rid="B36">Guyon and Elisseeff, 2003</xref>). Thus, some of them can be removed without the loss of information. The correlation can be positive or negative. In the positive case, both variables increase or decrease. Whereas in the case of negative correlation, when a variable increases the other decreases. In this study, the linear correlation between features is analyzed through the Pearson correlation coefficient, whose values are in the range of &#x2212;1 to 1. A correlation equal to &#x2212;1 corresponds to a perfect negative correlation, while &#x2b;1 is a perfect positive correlation. A coefficient equal to 0 means that there is no relationship between the features. The interpretation of the coefficient can be [0&#x2013;0.2), [0.2&#x2013;0.4), [0.4&#x2013;0.6), [0.6&#x2013;0.8) and [0.8&#x2013;1], which correspond to a very low, low, moderate, high and very high correlation, respectively. The negative correlation works in the same way but with a negative sign. Pearson&#x2019;s correlation is a frequent method for variable selection, which has been used for the prediction of soil classes (<xref ref-type="bibr" rid="B19">Camera et al., 2017</xref>; <xref ref-type="bibr" rid="B21">Campos et al., 2018</xref>), soil depth (<xref ref-type="bibr" rid="B19">Camera et al., 2017</xref>; <xref ref-type="bibr" rid="B56">Lu et al., 2019</xref>) and soil organic matter (<xref ref-type="bibr" rid="B23">Chen et al., 2022</xref>).</p>
</sec>
</sec>
<sec id="s2-6">
<title>2.6 Data pre-processing: training and validation points</title>
<p>In order to predict the occurrence of AS soils in the study area, the model must first be trained and validated with the soil samples and their corresponding values of the environmental covariates. The relationship between the soil samples and the values of the covariates allows the model to learn the characteristics of both classes during training. In this way, the model will be able to predict the AS soils from the values of the covariates. This study is a binary classification between AS and non-AS soils. For a good classification of both classes with machine learning techniques, it is important to have a balanced dataset with an equal number of samples of each class (<xref ref-type="bibr" rid="B84">Weiss and Provost, 2001</xref>; <xref ref-type="bibr" rid="B70">Porwal et al., 2003</xref>; <xref ref-type="bibr" rid="B83">Wei and Dunbrack, 2013</xref>). In this study, the dataset consists of 186 soil samples or cores, 93 for each class. The soil samples have been divided in two groups, the larger with 80% of the samples for training the model, and the smaller one with 20% of the samples for the validation. It should be noted that for a good performance of the model, both the training set and the validation set must also be balanced. Therefore, in the training dataset there are 148 soil samples, 74 for each class. Whereas, in the validation dataset there are 38 samples, 19 for each class. Inset of <xref ref-type="fig" rid="F1">Figure 1</xref> shows the soil samples in the study area. The same training and validation sets used in the previous work by (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>) have been considered in this study.</p>
</sec>
<sec id="s2-7">
<title>2.7 Metrics for evaluation</title>
<p>The metrics associated with the confusion matrix have been considered to evaluate the effectiveness of the models for the classification and prediction of the AS soil occurrence. These metrics give information about the classification and prediction of each class, which allows a better interpretation of the performance of the model in a binary classification. The metrics related to the confusion matrix are precision, recall and F1-score (<xref ref-type="bibr" rid="B71">Powers, 2011</xref>). The precision indicates the proportion of correctly predicted samples for a given class compared to the total number of predicted samples for that class. The recall is the percentage of samples properly classified for a given class. This metric also receives other names such as sensitivity, true positive rate, or hit rate (<xref ref-type="bibr" rid="B62">M&#xfc;ller and Guido, 2016</xref>). In order to avoid misinterpretation of the model performance, the precision and the recall have to be considered together. High values of both metrics for a given class show a good ability of the model to correctly predict and classify this class. On the other hand, the F1-score is a metric that merges the precision and the recall, which equation is<disp-formula id="e1">
<mml:math id="m3">
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>This metric is very important in binary classification and specially with unbalanced datasets, as it gives information about how the model works for each class. A high value of the F1-score, the closer to one the better, indicates that the model performs well for a given class.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and discussion</title>
<sec id="s3-1">
<title>3.1 Selected covariate groups and their evaluation for the classification of acid sulfate soils</title>
<p>In general, machine learning models perform better for large datasets. Thus, it is expected that increasing the number of the environmental covariates will contribute to a better classification of the AS soils. In a previous work, five environmental covariates (DEM, slope, quaternary, real and imaginary components of the aerogeophysics layer) were used for the classification and prediction of AS soils for the same soil samples (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). In this study, the initial raster dataset consists of 17 environmental covariates. First, the machine learning models, RF and GB, have been evaluated considering all covariates. In the case of RF, the consideration of the 17 covariates improves the results between 6%&#x2013;10% compared to the previous study with only five covariates (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). However, GB gives the same results, as shown in <xref ref-type="table" rid="T1">Table 1</xref>, where the metrics calculated from the confusion matrix are represented for both methods. This indicates that the consideration of all variables does not necessarily lead to better results. Therefore, the selection of the most relevant variables for an accurate classification of AS soils is essential. However, the variable selection is a complex task. In this work, we have made the selection of variables in three different ways, which allows a more precise selection. The first way is to select a fixed number of most important variables for a given method. As a result, the set of selected variables will perform well when they are together. It is important to note that an irrelevant feature by itself can improve the performance of the model when other features are considered (<xref ref-type="bibr" rid="B36">Guyon and Elisseeff, 2003</xref>). As the selection of variables depends on the method, different variable selection methods have been used. <xref ref-type="table" rid="T2">Table 2</xref> shows the environmental covariates selected by eight different methods: UFS, RF, GB, ETC, RFE &#x2b; RF, RFE &#x2b; GB, RFE &#x2b; ETC and RFE &#x2b; LR. The number of selected variables is limited to the ten most important for each method. From these results, the frequently selected covariates can be determined. As it can be seen, DEM is selected by all methods, while MRRTF is never selected. Thus, DEM is a very important variable for the characterization of AS soils in this case study, whereas MRRTF is irrelevant. Among the most frequently selected variables are the ten covariates selected by RF (<xref ref-type="table" rid="T2">Table 2</xref>). Furthermore, with this group of variables, both RF and GB obtain their best results. This can be seen in <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref> where the results of the models for the different covariate groups are shown for RF and GB, respectively. This indicates that RF is a very good method to select important variables for AS soils. On the contrary, the group of variables selected by ETC is the one that gives the worst results for both methods (<xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref>). Furthermore, it should be noted that for this group of covariates, the results for both methods are worse than the ones obtained in the previous study, where only five covariates were considered (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). Thus, ETC is not one of the best methods to select the most important environmental covariates for the classification of AS soils.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Metrics related to the confusion matrix for the case in which all environmental covariates are considered for Random Forest (RF) and Gradient Boosting (GB). The two classes are non acid sulfate (non-AS) and acid sulfate (AS) soils.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="left">Class</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">RF</td>
<td align="left">non-AS</td>
<td align="center">0.82</td>
<td align="center">0.74</td>
<td align="center">0.78</td>
</tr>
<tr>
<td align="left"/>
<td align="left">AS</td>
<td align="center">0.76</td>
<td align="center">0.84</td>
<td align="center">0.80</td>
</tr>
<tr>
<td align="left">GB</td>
<td align="left">non-AS</td>
<td align="center">0.78</td>
<td align="center">0.74</td>
<td align="center">0.76</td>
</tr>
<tr>
<td align="left"/>
<td align="left">AS</td>
<td align="center">0.75</td>
<td align="center">0.79</td>
<td align="center">0.77</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Environmental covariates selected by different variable selection methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Covariates</th>
<th align="center">UFS</th>
<th align="center">RF</th>
<th align="center">GB</th>
<th align="center">ETC</th>
<th align="center">RFE &#x2b; RF</th>
<th align="center">RFE &#x2b; GB</th>
<th align="center">RFE &#x2b; ETC</th>
<th align="center">RFE &#x2b; LR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">DEM</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">Slope</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">Quaternary</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">aem-real</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="left">
</td>
</tr>
<tr>
<td align="left">aem-im</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
</tr>
<tr>
<td align="left">aem-resist</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
</tr>
<tr>
<td align="left">Aspect</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
</tr>
<tr>
<td align="left">Hillshade</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
</tr>
<tr>
<td align="left">Roughness</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">Profilecur</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">tang-cur</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">
</td>
</tr>
<tr>
<td align="left">MRRTF</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
</tr>
<tr>
<td align="left">MRVBF</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="left"/>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">TPI</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">TRI</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">valley-dep</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
</tr>
<tr>
<td align="left">TWI</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
<td align="center">
</td>
<td align="center">
</td>
<td align="center">&#x2022;</td>
<td align="center">&#x2022;</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Variable selection methods: Univariate feature selection (UFS), Random Forest (RF), Gradient Boosting (GB), Extra Trees Classifier (ETC), and Recursive feature elimination (RFE) in combination with other techniques: RF, GB, ETC, and Logistic regression (LR). Covariates: DEM, digital elevation model; slope; quaternary; aem-real, aem-im, aem-resist: real, imaginary and apparent resistivity aeroelectromagnetic components; Aspect; Hillshade; Roughness; Profilecur, profile curvature; tang-cur, tangential curvature; MRRTF: multiresolution index of ridge top flatness; MRVBF, multiresolution index of valley bottom flatness; TPI, topographic position index; TRI, terrain ruggedness index; valley-dep, valley depth; TWI, topographic wetness index.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Metrics related to the confusion matrix for different groups of environmental covariates for Random Forest (RF). The classes are acid sulfate (AS) and non acid sulfate (non-AS) soils.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Covariates selected by</th>
<th align="left">Class</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">UFS</td>
<td align="left">non-AS</td>
<td align="center">0.82</td>
<td align="center">0.74</td>
<td align="center">0.78</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.76</td>
<td align="center">0.84</td>
<td align="center">0.80</td>
</tr>
<tr>
<td rowspan="2" align="left">RF, RFE &#x2b; RF</td>
<td align="left">non-AS</td>
<td align="center">0.88</td>
<td align="center">0.79</td>
<td align="center">0.83</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.81</td>
<td align="center">0.89</td>
<td align="center">0.85</td>
</tr>
<tr>
<td rowspan="2" align="left">GB, RFE &#x2b; ETC, RFE &#x2b; LG</td>
<td align="left">non-AS</td>
<td align="center">0.78</td>
<td align="center">0.74</td>
<td align="center">0.76</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.75</td>
<td align="center">0.79</td>
<td align="center">0.77</td>
</tr>
<tr>
<td rowspan="2" align="left">ETC</td>
<td align="left">non-AS</td>
<td align="center">0.67</td>
<td align="center">0.63</td>
<td align="center">0.65</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.65</td>
<td align="center">0.68</td>
<td align="center">0.67</td>
</tr>
<tr>
<td rowspan="2" align="left">RFE &#x2b; GB</td>
<td align="left">non-AS</td>
<td align="center">0.83</td>
<td align="center">0.79</td>
<td align="center">0.81</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.80</td>
<td align="center">0.84</td>
<td align="center">0.82</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Metrics related to the confusion matrix for different groups of environmental covariates for Gradient Boosting (GB). The classes are acid sulfate (AS) and non acid sulfate (non-AS) soils.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Covariates selected by</th>
<th align="left">Class</th>
<th align="center">Precision</th>
<th align="right">Recall</th>
<th align="right">F1-score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">UFS, RFE &#x2b; RF, RFE &#x2b; GB,</td>
<td align="left">non-AS</td>
<td align="center">0.78</td>
<td align="center">0.74</td>
<td align="center">0.76</td>
</tr>
<tr>
<td align="left">RFE &#x2b; ETC, RFE &#x2b; LR</td>
<td align="left">AS</td>
<td align="center">0.75</td>
<td align="center">0.79</td>
<td align="center">0.77</td>
</tr>
<tr>
<td rowspan="2" align="left">RF</td>
<td align="left">non-AS</td>
<td align="center">0.83</td>
<td align="center">0.79</td>
<td align="center">0.81</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.80</td>
<td align="center">0.84</td>
<td align="center">0.82</td>
</tr>
<tr>
<td rowspan="2" align="left">GB</td>
<td align="left">non-AS</td>
<td align="center">0.82</td>
<td align="center">0.74</td>
<td align="center">0.78</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.76</td>
<td align="center">0.84</td>
<td align="center">0.80</td>
</tr>
<tr>
<td rowspan="2" align="left">ETC</td>
<td align="left">non-AS</td>
<td align="center">0.72</td>
<td align="center">0.68</td>
<td align="center">0.70</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.70</td>
<td align="center">0.74</td>
<td align="center">0.72</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Another set of variables that gives very good results with RF is the one selected by RFE &#x2b; RF (<xref ref-type="table" rid="T3">Table 3</xref>). However, for this group the results do not improve for GB with respect to the study of five covariates (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). It should be noted that increasing the number of environmental covariates to ten improves the results of RF for all the groups except for the one selected by ETC (<xref ref-type="table" rid="T3">Table 3</xref>). On the contrary, in the case of GB, the increase of environmental covariates only improves the results for two cases, those selected by RF and GB (<xref ref-type="table" rid="T4">Table 4</xref>). Therefore, a greater number of variables generally benefits the classification in the case of RF but not necessarily when using GB. It could be related to the fact that unlike other methods, the consideration of irrelevant variables does not have a serious impact on the RF model (<xref ref-type="bibr" rid="B49">Kuhn and Johnson, 2013</xref>).</p>
<p>Contrary to the wrapped methods, the filter method UFS makes the selection based on the relationship between each environmental covariate and AS soils. Thus, the environmental covariates selected by this method give relevant information for the classification of AS soils. As it can be seen in <xref ref-type="table" rid="T2">Table 2</xref>, the covariates selected by UFS are the same as by RF except for one, roughness instead of quaternary. However, the results obtained for the group selected by UFS are not as good as for the one selected by RF, see <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref>. Even for the GB model the results for this group do not improve with respect to the previous study with five covariates (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). This shows that the selection of each variable, as well as the combination between them, is extremely important for an accurate prediction or classification of AS soils.</p>
<p>So far, we have made the selection taking into account the importance of the covariates for the given methods. However, the redundancy of the variables is also very important for their selection. A pertinent question is how the correlation between variables affects the performance of the model. The linear correlation between the 17 environmental covariates considered in this study is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. In the heatmap, dark green colors represent a strong positive correlation, whereas dark red colors indicate a strong negative correlation. A very low, low, moderate, high and very high correlation correspond to [0&#x2013;0.2), [0.2&#x2013;0.4), [0.4&#x2013;0.6), [0.6&#x2013;0.8) and [0.8&#x2013;1], respectively. In general, variables with a high correlation provide redundant information, which can hinder the prediction. This could be the case of the group selected by RFE &#x2b; LG, where some variables are strongly correlated with values close to one such as slope, roughness and TRI or profile curvature and TPI. For both, RF and GB models, the results obtained with this group are similar to the results by (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>) with five covariates and GB model. However, there are other covariates that are highly correlated and together give good results as in the case of the real and imaginary components of the aerogeophysics layers. Although these covariates are correlated, they also provide different information that can help to localize AS soils. The real component may show deep bedrock anomalies related to sulfide deposits, while the imaginary component allows the identification of weak surface anomalies associated to variations in the thickness of the topsoil. In order to analyze the role that the correlation is playing in the performance of the models, we have compared the correlation of two of the groups of variables. The two groups considered are those that have given the best and worst results. Curiously, the group with the best results, the one selected by RF, has more correlation than the one selected by ETC. Contrary to what might be expected, the group that shows the poorest results in the classification of AS soils has a very low correlation. Thus, a low correlation between the covariates does not guarantee the best prediction of the model. Comparing both groups, it can be seen that they only differ in three covariates, slope, real component of the aerogeophysics layer and valley depth in the group selected by RF, and aspect, profile curvature and tangential curvature in the ETC one (<xref ref-type="table" rid="T2">Table 2</xref>). Therefore, it seems that slope, real component of the aerogeophysics layer and valley depth are more important for the classification of AS soils than aspect, profile curvature or tangential curvature. Looking at <xref ref-type="table" rid="T2">Tables 2</xref>, <xref ref-type="table" rid="T3">3</xref> and <xref ref-type="table" rid="T4">Table 4</xref>, it is verified that the best results are obtained for the groups where these three covariates are presented. Moreover, valley depth is selected by 75% of the methods as one of the most important, slope and real component of the aerogeophysics layer by 62.5%, whereas aspect, profile curvature and tangential curvature are only selected by 37.5% of the methods. On the other hand, the correlation between the covariates selected by RFE &#x2b; RF is very high. For example, slope, roughness, and TRI are strongly correlated with values close to one, and real and imaginary components of the aerogeophysics layer are highly correlated (<xref ref-type="fig" rid="F2">Figure 2</xref>). In addition, there are several cases with moderate correlation. Despite the high correlation, the RF model also gives the best results for this group, but this is not the case for GB (<xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref>). This could mean that GB is more affected by the correlation. By removing the slope and roughness of the RFE &#x2b; RF group, the correlation related to the terrain layers disappears, and the results for both models improve. In the case of RF, the different metrics related to the confusion matrix increase between 1%&#x2013;5%, leading to the best results of this model (<xref ref-type="table" rid="T5">Table 5</xref>). For GB, all the metrics improve by 5%, matching the best results obtained with this model for the group selected by RF (<xref ref-type="table" rid="T4">Table 4</xref>). Thus, for a good performance of the model there must be a balance between the importance of the covariates and their correlation.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>(Color online) Heatmap of Pearson correlation coefficient between the environmental covariates considered in the study.</p>
</caption>
<graphic xlink:href="fenvs-11-1213069-g002.tif"/>
</fig>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Metrics related to the confusion matrix for the case RFE &#x2b; RF without the correlation of the terrain layers and for the group selected by hand considering only the correlation for Random Forest (RF) model. The classes are acid sulfate (AS) and non acid sulfate (non-AS) soils.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="left">Class</th>
<th align="center">Precision</th>
<th align="center">Recall</th>
<th align="center">F1-score</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="left">RF</td>
<td align="left">non-AS</td>
<td align="center">0.89</td>
<td align="center">0.84</td>
<td align="center">0.86</td>
</tr>
<tr>
<td align="left">AS</td>
<td align="center">0.85</td>
<td align="center">0.89</td>
<td align="center">0.87</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In this work, the second way of variable selection has been carried out considering only the correlation. The selection has been made avoiding the high correlation between the covariates. If some variables have a correlation larger than 0.5 only one of them is selected. This allows to introduce more information in the model but without redundancy. All possible combinations have been analyzed, leading to different groups with eleven variables. It is worth highlighting the group formed by the following covariates: DEM, quaternary, imaginary and apparent resistivity components of the aerogeophysics layer, aspect, hillshade, tangential curvature, MRVBF, TRI, valley depth and TWI. For this group, RF also gives the best results, the same ones as for the case selected by RFE &#x2b; RF but without slope and roughness (<xref ref-type="table" rid="T5">Table 5</xref>). For GB, the results for this group are similar to the results obtained with the group selected by RF except for the recall, which increases by 5% for AS soils but decreases by 5% for non-AS soils (<xref ref-type="table" rid="T4">Table 4</xref>). This once again demonstrates the complexity of variable selection.</p>
<p>The Backward Selection is the third method considered for variable selection in this study. As it was already explained in <xref ref-type="sec" rid="s2-5-2-2">subsection 2.5.2.2</xref>, the method is based on the elimination of the irrelevant features for the model. Starting from the entire set of environmental covariates, the importance of each variable for the model is measured, and the most irrelevant one is removed. This process is repeated until the performance of the model stops improving. The importance of the variables is given by the model. In this study, Backward Selection has been analyzed using RF and GB. In the case of RF, the first covariate eliminated is MRRTF, which improves the results between 0%&#x2013;5%. The second feature eliminated is the roughness, which leads to equal the results obtained by the groups selected by RF and RFE &#x2b; RF (<xref ref-type="table" rid="T3">Table 3</xref>). However, removing the next least relevant covariate, MRVBF, the results are worse. For the GB model, the elimination of the first four irrelevant variables (MRVBF, TRI, MRRTF and slope) does not affect the results. By removing TPI, the results improve by 5%, matching the best results for this model, the ones obtained for the group selected by RF (<xref ref-type="table" rid="T4">Table 4</xref>). But by eliminating the following most irrelevant variable, aspect, the results are slightly worse. This indicates that the one-by-one backward selection method may select a set of variables that is not the best one for the performance of the model. This can be clearly seen in the case of the RF model, where the best results are obtained with 15 covariates. However, the same or better results are achieved with a smaller number of variables selected by other methods. The main problem with this backward selection is that the importance of the variables depends on the relationship between the variables considered, which changes as the variables are eliminated in each step. As already mentioned, an irrelevant variable by itself can be critical for a good performance of the model if other variables are present. Therefore, removing one of the variables can change the importance of certain variables. As a result, the selection of variables with this method is quite complex and does not guarantee that the variables selected are the most suitable for the model. It should be noted the difference between this one-by-one backward selection and the backward selection of the RFE method, where all variables are selected at the same time and are relevant to the model when they are together.</p>
<p>From all these results it can be seen that the environmental covariates DEM, slope, quaternary, the three components of the aerogephysics layers, hillshade, MRVBF, valley depth and TWI are very important for the classification of AS soils. In this study, the consideration of these covariates leads to a very accurate classification and prediction of this type of soil for the two models analyzed, RF and GB. In general, for the same groups of covariates better results are obtained with RF than with GB. It should be noted that the RF model improves the accuracies between 15%&#x2013;17% for the case of eight relevant covariates with respect to a previous study where five covariates were considered (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). Unlike RF, the results for most of the groups analyzed with the GB model do not improve with the increase of the number of covariates. Furthermore, in the cases in which the results improve with respect to the case of five covariates, the accuracies improve by at most 5%. In the study by (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>), the accuracies obtained with the GB model are 5%&#x2013;6% higher than those obtained with the RF model. This leads to two pertinent questions: Is GB a model that works better for a small number of covariates? or are these results related to the importance of the covariates considered? In order to answer these questions, a new group of five covariates has been analyzed. In the case of RF, there are six covariates that appear in all the groups with the best results, i.e., with values of F1-score equal or larger than 0.83 and 0.85 for non-AS and AS soils, respectively (<xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T5">5</xref>). These environmental covariates are: DEM, quaternary, imaginary and apparent resistivity components of the aerogeophysics layer, hillshade and valley depth. As the quaternary layer is very relevant for GB, this layer is not included in the study. For this set of variables, the RF model improves the classification between 5%&#x2013;6% with respect to the results obtained in a previous work where the five covariates considered were DEM, slope, quaternary, real and imaginary components of the aerogeophysics layer (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). For GB, the results obtained for this set of variables are worse than the ones for the previous work. This is because the quaternary layer, which is very important for GB model, is not included in this study. Thus, this shows the great importance of the selection of features in the classification and prediction of AS soils for each model.</p>
</sec>
<sec id="s3-2">
<title>3.2 Probability map for acid sulfate soil occurrences</title>
<p>In general the accuracies obtained in this study are greater for RF than for GB. Therefore, RF is the model chosen for the mapping of AS soils. The group of covariates from which the best results have been obtained is the one used to create the AS soil probability map. There are two sets of variables that have given the best results with the RF model (<xref ref-type="table" rid="T5">Table 5</xref>), one where only the correlation was taken into account in the selection, and the other the one selected by RFE &#x2b; RF without the correlation between terrain layers. The first one has eleven covariates whereas the second one eight covariates. In general, smaller datasets reduce computation time (<xref ref-type="bibr" rid="B36">Guyon and Elisseeff, 2003</xref>). Although in our case there will be hardly any difference between the two groups, the smaller one has been used. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the AS soil probability map created with RF for the group composed of the environmental covariates: DEM, quaternary, the real, imaginary and apparent resistivity components of the aerogephysics layer, hillshade, TRI and valley depth. The prediction of the probability of encountering AS soils has been performed for each of the 434,036 cells of 50 m &#xd7; 50 m that make up the study area. The calculation is based on the values of the environmental covariates. For the representation of the map, the probability of encountering AS soils has been divided into four different probability classes: very low, low, high and very high, which correspond to [0&#x2013;0.25), [0.25&#x2013;0.5), [0.5&#x2013;0.75) and [0.75&#x2013;1], respectively. The corresponding areas of the AS soil probability map in percentage and km<sup>2</sup> for each probability class are represented in <xref ref-type="table" rid="T6">Table 6</xref>. For 21% of the study area the probability of encountering AS soils is high and very high, whereas in the remaining 79% the probability is low and very low. As has already been mentioned, the RF model for this group of covariates increases the accuracies with respect to the previous study (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>), between 15%&#x2013;17% for the RF model and between 10%&#x2013;11% for the GB model. Comparing the AS soil probability maps created with the RF model in both cases, it can be seen that the areas with very high, high and very low probabilities are smaller in the present map, the one with greater accuracies. On the contrary, the area with low probability has increased to represent about half (51%) of the study area (<xref ref-type="table" rid="T6">Table 6</xref>). It should be noted that approximately 58% of the uppermost meter in the study area is made up of bedrock, outcrops and block fields, where the probability of encountering AS soils is usually very low to low. Although the model has not been trained with samples from the areas where bedrock, outcrops and rockfields are located, the model has predicted around 86% of these areas as very low and low probability. This demonstrates a high ability of the RF model to predict and classify AS soils with this set of environmental covariates.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>(Color online) Probability map created from Random Forest (RF) for the covariate group selected by RFE &#x2b; RF without slope and roughness.</p>
</caption>
<graphic xlink:href="fenvs-11-1213069-g003.tif"/>
</fig>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Validation of the probability map created from Random Forest (RF) for the group of covariates: DEM, quaternary, aem-real, aem-im, aem-resist, hillshade, TRI and valley depth. Validation points are acid sulfate (AS) and non-acid sulfate (non-AS) soils.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="left">
<bold>Probability zone</bold>
</th>
<th rowspan="2" align="center">
<bold>% Of study area</bold>
</th>
<th rowspan="2" align="center">
<bold>km</bold>
<sup>
<bold>2</bold>
</sup> <bold>of study area</bold>
</th>
<th colspan="2" align="center">
<bold>Validation points</bold>
</th>
</tr>
<tr>
<th align="center">
<bold>AS</bold>
</th>
<th align="center">
<bold>non-AS</bold>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Very high [0.75&#x2013;1]</td>
<td align="center">5</td>
<td align="center">44</td>
<td align="center">11</td>
<td align="center">1</td>
</tr>
<tr>
<td align="left">High [0.5&#x2013;0.75)</td>
<td align="center">16</td>
<td align="center">177</td>
<td align="center">6</td>
<td align="center">2</td>
</tr>
<tr>
<td align="left">Low [0.25&#x2013;0.5)</td>
<td align="center">51</td>
<td align="center">556</td>
<td align="center">2</td>
<td align="center">8</td>
</tr>
<tr>
<td align="left">Very low [0&#x2013;0.25)</td>
<td align="center">28</td>
<td align="center">297</td>
<td align="center">0</td>
<td align="center">8</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>On the other hand, there is a linear feature in the probability map (<xref ref-type="fig" rid="F3">Figure 3</xref>), which is related to a visible power line in the aerogeophysics layers. These layers show the components of the electromagnetic induction, which is very strong in the power line. Thus, the information of the area of the line is related to the power line but not to the soil. This can affect the prediction of the model for the area of the power line depending on the importance of the different covariates. If the aerogeophysics layers are very important for the model, the prediction in the area of the power line may be incorrect. A more accurate prediction will be made if some of the covariates that give soil information in the power line area are the most relevant for the model. Unlike the line feature shown in the AS soil probability maps created with different models in the previous study (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>), the line feature is not as sharp on this map, i.e., the predictions for this line are much more similar to the predictions of the neighboring cells (<xref ref-type="fig" rid="F3">Figure 3</xref>). This is due to the number of covariates that give information about the soil in the area of the power line is larger in this study, which facilitates a more accurate prediction. In order to avoid possible linear features in the maps in future studies, it should be good to mask any power lines present in the aerogeophysics layers before doing the prediction for the area.</p>
<p>Finally, the validation of the AS soil probability map can be performed by comparing the validation points to the prediction made by the model for the cells where these points are located. These results are shown in <xref ref-type="table" rid="T6">Table 6</xref>. For this validation, the same points used to evaluate the models for the different groups of environmental covariates have been considered. These validation points are displayed in the probability map (<xref ref-type="fig" rid="F3">Figure 3</xref>). For the AS soil validation points, 89% of the predicted probability classes for the corresponding cells are correctly classified. Most of these points (65%) are located in areas predicted as very high probability areas, whereas the remaining points (35%) in high probability areas. There are two points located in cells that have been predicted as belonging to the low probability class. For the case of non-AS soils validation points, 84% of their corresponding cells are correctly classified in the low and very low probability areas in equal proportion for both areas (<xref ref-type="table" rid="T6">Table 6</xref>). There are three validation points (16%) located in cells that have been incorrectly classified, two points are in the high probability area and one point in the very high probability area.</p>
</sec>
<sec id="s3-3">
<title>3.3 Extents of acid sulfate soil areas</title>
<p>The main objective of mapping the occurrence of the AS soils is to locate areas that could have a negative environmental impact if the soil materials are disturbed for instance in agriculture and during infrastructure developments. Furthermore, it is important to know the extent of AS soils in order to estimate potential or ongoing mobilization of environmentally hazardous elements (e.g., Cd and Ni) into watercourses. So far, the extent of AS soils has been calculated from conventional occurrence maps. In the study area, a conventional probability map of AS soil occurrence within the Littorina Sea maximum extent area (83% of the study area) was presented in (<xref ref-type="bibr" rid="B26">Est&#xe9;vez et al., 2022</xref>). This conventional AS soil probability map has four different probability classes: high, moderate, low and very low, which probabilities of encountering AS soils are 98.5%, 52.5%, 1.7% and 0%, respectively. The total area of the conventional map is 904.48 km<sup>2</sup>, where 24.65&#xa0;km<sup>2</sup> corresponds to the high class, 78.88&#xa0;km<sup>2</sup> to the moderate class, 207.72&#xa0;km<sup>2</sup> to the low class and 593.23&#xa0;km<sup>2</sup> to the very low class. In this paper, the extent of AS soils for this conventional map has been calculated based on a similar approach previously used for calculating the extent of AS soils in Denmark (<xref ref-type="bibr" rid="B57">Madsen and Jensen, 1988</xref>). In total, the study area comprises 69.29&#xa0;km<sup>2</sup> AS soils (7.7% of the area) of which 24.28&#xa0;km<sup>2</sup> are present in the high probability class, 41.43&#xa0;km<sup>2</sup> in the moderate probability class and 3.58&#xa0;km<sup>2</sup> in the low probability class. The distribution of roughly 8% of AS soils in the study area is considerably somewhat lower compared to a larger area of 3,106&#xa0;km<sup>2</sup> located in Northern Ostrobothnia, Finland, where about 25% of the total area is covered by AS soils (<xref ref-type="bibr" rid="B7">Becher et al., 2019</xref>). This discrepancy is most likely due to differences in topography between the two regions, with much more variation in the study area and with a general abundance of bedrock outcrops and a lack of larger rivers feeding sedimentary basins with sediment and organic matter required for iron sulfide formation in southern Finland. However, it should be noted that almost 60% of the study area described in this paper is covered by bedrock, outcrops and block fields that could not be sampled. In the conventional AS soil occurrence map, the probability of finding AS soils in an area is determined from the proportion of soil samples classified as AS soils and the total number of soil samples of the area. Thus, the probability of the 60% of the study area is equal to zero as there are not soil samples in that area. This may affect the estimation of the extent of AS soils in the conventional AS soil probability map.</p>
<p>In this paper, a new approach for the calculation of the extent of AS soils in modeled AS soil probability maps is shown. Unlike in the conventional map, the probability of encountering AS soils in a modeled probability map has been calculated for each pixel of the map (50&#xa0;m &#xd7; 50&#xa0;m). Therefore, the extent of AS soils can be calculated by multiplying each pixel area by its probability of finding AS soils. This allows a much more accurate calculation of the extent of AS soils. The extent of AS soils in the modeled probability map has been calculated for the same area as the conventional map, the part corresponding to the Littorina Sea maximum extent. The total extent of AS soils in the modeled map is 315.5 km<sup>2</sup>, of which 42.7&#xa0;km<sup>2</sup> are located in the very high probability area, 87.2&#xa0;km<sup>2</sup> in the high probability area, 151.5&#xa0;km<sup>2</sup> in the low probability area and 34.1&#xa0;km<sup>2</sup> in the very low probability area. The total extent of AS soils represents 35% of the study area. This value must be interpreted as the maximum extent of AS soils as all cells of the map have been considered. For the potential environmental hazards, it should be noted that only 129.9&#xa0;km<sup>2</sup> of the calculated extents of AS soils have a probability of encountering AS soils greater than 50%.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>As it has been shown in this study, considering a larger number of environmental covariates does not necessarily improve the prediction of the model. For this reason, in this study, variable selection has been used to improve the prediction accuracy for acid sulfate soil mapping. Eleven different variable selection methods have been considered for the selection of the most relevant environmental covariates for the correct classification and prediction of acid sulfate (AS) soils. Among the most frequently chosen variables are those selected by Random Forest (RF). Furthermore, the best results for the two modeling methods, RF and Gradient Boosting (GB), have been obtained for the group of covariates selected by RF. Therefore, in this case study, RF is a very good method in the selection of environmental covariates for the prediction of AS soils. On the contrary, Extra Trees Classifier (ETC) is the method whose selection of environmental covariates has led to the worst results for both modeling methods.</p>
<p>On the other hand, it has been seen that the combination of two variable selection methods can improve the prediction accuracy. This has been the case when considering RFE &#x2b; RF and Pearson&#x2019;s correlation, where the results have improved between 1%&#x2013;5%, if the correlation is taken into account. However, it has been seen that the correlation alone is not enough to select the most important covariates for the prediction of AS soils. For instance, there are strongly correlated covariates such as the real and imaginary components of the aerogeophysical layers that combined have given very good results in the prediction. Others covariates, such as highly correlated terrain layers, have made prediction difficult. Furthermore, it has been shown that a group of covariates without correlation does not necessarily give a good prediction. Finally, the results of the Backward Selection analysis have shown that this method does not generally select the most appropriate covariates for a correct model prediction.</p>
<p>In general, better results have been obtained in the prediction of AS soils with the RF model than with the GB model. The AS soil probability map has been created for the group of covariates with the best results in prediction for the RF model. Variable selection has enabled the RF model to improve the results of the prediction by up to 15%&#x2013;17% for a group of eight covariates as compared with a previous study in the same area where five covariates were considered. It should be noted that this improvement is not produced by the increase of three extra covariates in the study, but by the consideration of eight covariates in particular. For eight different layers, that improvement does not occur. This demonstrates the importance of variable selection for the prediction of AS soils. From the validation of the AS soil probability map, it can be seen that the model has been able to correctly predict 89% of the cells where the validation points are located for AS soils, and 84% for non-AS soils. Finally, this study presents a new approach that allows an accurate estimation of the extent of AS soils in modeled probability maps.</p>
<p>Future studies should address the importance of these selected environmental covariates for classification and prediction of AS soils in other areas where AS soils may be slightly different. Another important study would be the analysis of the relevance of these environmental covariates for machine learning methods with a very different algorithm, such as a Convolutional Neural Network.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The raw data supporting the conclusion of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>VE carried out the research, performed the analysis and modeling, and wrote the paper. SM and ABo contributed to write the paper. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This work has been financially supported by Stiftelsen Arcada foundation (Finland).</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Airo</surname>
<given-names>M-L.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Aerogephysics in Finland 1972-2004 methods, system characteristics and applications</article-title>. <source>Geol. Surv. Finl.</source>, <fpage>197</fpage>. <comment>Special Paper 39. Espoo, Finland</comment>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Airo</surname>
<given-names>M.-L.</given-names>
</name>
<name>
<surname>Loukola-Ruskeeniemi</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Characterization of sulfide deposits by airborne magnetic and gamma-ray responses in eastern Finland</article-title>. <source>Ore Geol. Rev.</source> <volume>24</volume>, <fpage>67</fpage>&#x2013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1016/j.oregeorev.2003.08.008</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Akusok</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bj&#xf6;rk</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Est&#xe9;vez</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Boman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Randomized model structure selection approach for Extreme learning machine applied to acid sulfate soil detection</article-title>,&#x201d; in <source>Proceedings of ELM 2021. ELM 2021. Proceedings in adaptation, learning and optimization</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Bj&#xf6;rk</surname>
<given-names>K. M.</given-names>
</name>
</person-group> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>). <pub-id pub-id-type="doi">10.1007/978-3-031-21678-7_4</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#xc5;str&#xf6;m</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bj&#xf6;rklund</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Geochemistry and acidity of sulphide-bearing postglacial sediments of Western Finland</article-title>. <source>Environ. Geochem. Health</source> <volume>19</volume>, <fpage>155</fpage>&#x2013;<lpage>164</lpage>. <pub-id pub-id-type="doi">10.1023/a:1018462824486</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azizi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ayoubi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nabiollahi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Garosi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gislum</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Predicting heavy metal contents by applying machine learning approaches and environmental covariates in west of Iran</article-title>. <source>J. Geochem. Explor.</source> <volume>233</volume>, <fpage>106921</fpage>. <pub-id pub-id-type="doi">10.1016/j.gexplo.2021.106921</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Becher</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sohlenius</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>&#xd6;hrling</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Boman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Josefsson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mattb&#xe4;ck</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). &#x201c;<article-title>Acid sulphate soils around coastal watercourses, Project report</article-title>,&#x201d; in <source>2019, coastal watercourses - methodological development and restoration. Final report</source>, <fpage>189</fpage>. <comment>Interreg Nord 2014-2020</comment>.</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Becher</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sohlenius</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>&#xd6;hrling</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Boman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Josefsson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mattb&#xe4;ck</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Sur sulfatjord runt kustmynnande vattendrag. Technical report</source>. <publisher-loc>Uppsala, Sweden</publisher-loc>: <publisher-name>Geological Survey of Sweden and Geological Survey of Finland</publisher-name>, <fpage>35</fpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Behrens</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>A. X.</given-names>
</name>
<name>
<surname>Scholten</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>The ConMap approach for terrain-based digital soil mapping</article-title>. <source>Eur. J. Soil Sci.</source> <volume>61</volume>, <fpage>133</fpage>&#x2013;<lpage>143</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-2389.2009.01205.x</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beucher</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rasmussen</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Moeslund</surname>
<given-names>T. B.</given-names>
</name>
<name>
<surname>Greve</surname>
<given-names>M. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Interpretation of convolutional neural networks for acid sulfate soil classification</article-title>. <source>Front. Environ. Sci.</source> <volume>9</volume>, <fpage>809995</fpage>. <pub-id pub-id-type="doi">10.3389/fenvs.2021.809995</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beucher</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Adhikari</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Breuning-Madsen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Greve</surname>
<given-names>M. B.</given-names>
</name>
<name>
<surname>&#xd6;sterholm</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fr&#xf6;jd&#xf6;</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Mapping potential acid sulfate soils in Denmark using legacy data and LiDAR-based derivatives</article-title>. <source>Geoderma</source> <volume>308</volume>, <fpage>363</fpage>&#x2013;<lpage>372</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2016.06.001</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beucher</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Fr&#xf6;j&#xf6;</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>&#xd6;sterholm</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Martinkauppi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ed&#xe9;n</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Fuzzy logic for acid sulfate soil mapping: Application to the southern part of the Finnish coastal areas</article-title>. <source>Geoderma</source> <volume>226-227</volume>, <fpage>21</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2014.03.004</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beucher</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>&#xd6;sterholm</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Martinkauppi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ed&#xe9;n</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Fr&#xf6;j&#xf6;</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Artificial neural network for acid sulfate soil mapping: Application to the Sirppujoki River catchment area, south-Western Finland</article-title>. <source>J. Geochem Explor</source> <volume>125</volume>, <fpage>46</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1016/j.gexplo.2012.11.002</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Beucher</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Siemssen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Fr&#xf6;j&#xf6;</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>&#xd6;sterholm</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Martinkauppi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ed&#xe9;n</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Artificial neural network for mapping and characterization of acid sulfate soils: Application to Sirppujoki River catchment, southwestern Finland</article-title>. <source>Geoderma</source> <volume>247-248</volume>, <fpage>38</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2014.11.031</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Boman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Becher</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mattb&#xe4;ck</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sohlenius</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Auri</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>&#xd6;hrling</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Classification of acid sulphate soils in Finland and Sweden. Appendix 1</article-title>, p. In, <source>Coastal watercourses - methodological development and restoration</source>. <comment>Final report, Interreg Nord 2014-2020, 189 p. Available at: <ext-link ext-link-type="uri" xlink:href="https://www.lansstyrelsen.se/norrbotten/tjanster/publikationer/coastal-watercourses&#x2014;methodological-development-and-restoration.html">https://www.lansstyrelsen.se/norrbotten/tjanster/publikationer/coastal-watercourses&#x2014;methodological-development-and-restoration.html</ext-link>
</comment>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/a:1010933404324</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brungard</surname>
<given-names>C. W.</given-names>
</name>
<name>
<surname>Boettinger</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Duniway</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Wills</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Edwards</surname>
<given-names>T. C.</given-names>
<suffix>Jr.</suffix>
</name>
</person-group> (<year>2015</year>). <article-title>Machine learning for predicting soil classes in three semi-arid landscapes</article-title>. <source>Geoderma</source> <volume>239-240</volume>, <fpage>68</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2014.09.019</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brus</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Kempen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Heuvelink</surname>
<given-names>G. B. M.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Sampling for validation of digital soil maps</article-title>. <source>Eur. J. Soil Sci.</source> <volume>62</volume>, <fpage>394</fpage>&#x2013;<lpage>407</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-2389.2011.01364.x</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Camera</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zomeni</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Noller</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Zissimos</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Christoforou</surname>
<given-names>I. C.</given-names>
</name>
<name>
<surname>Bruggeman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A high resolution map of soil types and physical properties for Cyprus: A digital soil mapping optimization</article-title>. <source>Geoderma</source> <volume>285</volume>, <fpage>35</fpage>&#x2013;<lpage>49</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2016.09.019</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campling</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gobin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Feyen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Logistic modeling to spatially predict the probability of soil drainage classes</article-title>. <source>Soil Sci. Soc. Am. J.</source> <volume>66</volume>, <fpage>1390</fpage>&#x2013;<lpage>1401</lpage>. <pub-id pub-id-type="doi">10.2136/sssaj2002.1390</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Campos</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Giasson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Costa</surname>
<given-names>J. J. F.</given-names>
</name>
<name>
<surname>Machado</surname>
<given-names>I. R.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>E. B.</given-names>
</name>
<name>
<surname>Bonfatti</surname>
<given-names>B. R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Selection of environmental covariates for classifier training applied in digital soil mapping</article-title>. <source>Rev. Bras. Cienc. Solo.</source> <volume>42</volume>, <fpage>e0170414</fpage>. <pub-id pub-id-type="doi">10.1590/18069657rbcs20170414</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Castro Franco</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Domenech</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Costa</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Aparicio</surname>
<given-names>V. C.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Modelling effective soil depth at field scale from soil sensors and geomorphometric indices</article-title>. <source>Acta Agron&#xf3;mica.</source> <volume>66</volume> (<issue>2</issue>), <fpage>227</fpage>&#x2013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.15446/acag.v66n2.53282</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Comparison of feature selection methods for mapping soil organic matter in subtropical restored forests</article-title>. <source>Ecol. Indic.</source> <volume>135</volume>, <fpage>108545</fpage>. <pub-id pub-id-type="doi">10.1016/j.ecolind.2022.108545</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Debella-Gilo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Etzelm&#xfc;ller</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Spatial prediction of soil classes using digital terrain analysis and multinomial logistic regression modeling integrated in GIS: Examples from Vestfold County, Norway</article-title>. <source>Catena</source> <volume>77</volume>, <fpage>8</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1016/j.catena.2008.12.001</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Est&#xe9;vez Nu&#xf1;o</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Machine learning methods for classification of acid sulfate soils in Virolahti</article-title>. <source>Master&#x2019;s thesis</source>. <publisher-loc>Finland</publisher-loc>: <publisher-name>Arcada University of Applied Sciences</publisher-name>. <comment>Jan-Magnus Janssons plats 1, 00560 Helsinki, Finland (June 2020)</comment>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Est&#xe9;vez</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Beucher</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mattb&#xe4;ck</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Boman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bj&#xf6;rk</surname>
<given-names>K-M.</given-names>
</name>
<name>
<surname>Osterh&#xf6;lm</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Machine learning techniques for acid sulfate soil mapping in southeastern Finland</article-title>. <source>Geoderma</source> <volume>406</volume>, <fpage>115446</fpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2021.115446</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Est&#xe9;vez</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Mattb&#xe4;ck</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bj&#xf6;rk</surname>
<given-names>K-M.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Importance of the activation function in Extreme Learning Machine for Acid sulfate soil classification</source>. <comment>Presented at ELM 2022 &#x2013; Dec 8-9, 2022, Virtual Conference (Main location Helsinki &#x2013; Finland)</comment>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fitzpatrick</surname>
<given-names>B. R.</given-names>
</name>
<name>
<surname>Lamb</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Mengersen</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Ultrahigh dimensional variable selection for interpolation of point referenced spatial data: A digital soil mapping case study</article-title>. <source>PLoS ONE</source> <volume>11</volume> (<issue>9</issue>), <fpage>e0162489</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0162489</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Forman</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>An extensive empirical study of feature selection metrics for text classification</article-title>. <source>J. Mach. Learn. Res.</source> <volume>3</volume>, <fpage>1289</fpage>&#x2013;<lpage>1305</lpage>. <pub-id pub-id-type="doi">10.1162/153244303322753670</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Greedy function approximation: A gradient boosting machine</article-title>. <source>Ann. Stat.</source> <volume>29</volume> (<issue>5</issue>), <fpage>1189</fpage>&#x2013;<lpage>1232</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1013203451</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gambill</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Wall</surname>
<given-names>W. A.</given-names>
</name>
<name>
<surname>Fulton</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Howard</surname>
<given-names>H. R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Predicting USCS soil classification from soil property variables using Random Forest</article-title>. <source>J. Terramechanics</source> <volume>65</volume>, <fpage>85</fpage>&#x2013;<lpage>92</lpage>. <pub-id pub-id-type="doi">10.1016/j.jterra.2016.03.006</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<collab>Geological Survey of Finland</collab> (<year>2021</year>). <source>Acid sulfate soils &#x2013; map services</source>. <publisher-loc>Finland</publisher-loc>: <publisher-name>Geological Survey of Finland</publisher-name>. <ext-link ext-link-type="uri" xlink:href="http://gtkdata.gtk.fi/hasu/index.html">http://gtkdata.gtk.fi/hasu/index.html</ext-link>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geurts</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ernst</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wehenkel</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Extremely randomized trees</article-title>. <source>Mach. Learn.</source> <volume>1</volume>, <fpage>3</fpage>&#x2013;<lpage>42</lpage>. <pub-id pub-id-type="doi">10.1007/s10994-006-6226-1</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Giasson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Clarke</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>Vasconcellos Inda Junior</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Merten</surname>
<given-names>G. H.</given-names>
</name>
<name>
<surname>Tornquist</surname>
<given-names>C. G.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Digital soil mapping using multiple logistic regression on terrain parameters in southern Brazil</article-title>. <source>Sci. Agric. (Piracicaba, Braz.)</source> <volume>63</volume> (<issue>3</issue>), <fpage>262</fpage>&#x2013;<lpage>268</lpage>. <pub-id pub-id-type="doi">10.1590/s0103-90162006000300008</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grimm</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Behrens</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Maerker</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Elsenbeer</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Soil organic carbon concentrations and stocks on Barro Colorado Island &#x2014; digital soil mapping using Random Forests analysis</article-title>. <source>Geoderma</source> <volume>146</volume>, <fpage>102</fpage>&#x2013;<lpage>113</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2008.05.008</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guyon</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Elisseeff</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>An introduction to variable and feature selection</article-title>. <source>J. Mach. Learn. Res.</source> <volume>3</volume>, <fpage>1157</fpage>&#x2013;<lpage>1182</lpage>. <pub-id pub-id-type="doi">10.1162/153244303322753616</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guyon</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Weston</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Barnhill</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vapnik</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Gene selection for cancer classification using Support vector machines</article-title>. <source>Mach. Learn.</source> <volume>46</volume>, <fpage>389</fpage>&#x2013;<lpage>422</lpage>. <pub-id pub-id-type="doi">10.1023/a:1012487302797</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Haavisto-Hyv&#xe4;rinen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kutvonen</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2007</year>). <source>Maaper&#xe4;kartan k&#xe4;ytt&#xf6;opas</source>. <publisher-loc>Finland</publisher-loc>: <publisher-name>Geological Survey of Finland</publisher-name>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hall</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Holmes</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Benchmarking attribute selection techniques for discrete class data mining</article-title>. <source>IEEE Trans. Knowl. Data Eng.</source> <volume>15</volume> (<issue>6</issue>), <fpage>1437</fpage>&#x2013;<lpage>1447</lpage>. <pub-id pub-id-type="doi">10.1109/TKDE.2003.1245283</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hengl</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Mendes de Jesus</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Heuvelink</surname>
<given-names>G. B. M.</given-names>
</name>
<name>
<surname>Ruiperez Gonzalez</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kilibarda</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Blagoti&#x107;</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>SoilGrids250m: Global gridded soil information based on machine learning</article-title>. <source>PLoS One</source> <volume>12</volume>, <fpage>e0169748</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0169748</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heung</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bulmer</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Schimdt</surname>
<given-names>M. G.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Predictive soil parent material mapping at a regional-scale: A random forest approach</article-title>. <source>Geoderma</source> <volume>214-215</volume>, <fpage>141</fpage>&#x2013;<lpage>154</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2013.09.016</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heung</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>H. C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Knudby</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bulmer</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Schimdt</surname>
<given-names>M. G.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>An overview and comparison of machine-learning techniques for classification purposes in digital soil mapping</article-title>. <source>Geoderma</source> <volume>265</volume>, <fpage>62</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2015.11.014</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Nhan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>V. N. L.</given-names>
</name>
<name>
<surname>Johnston</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Lark</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Triantafilis</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Digital soil mapping of a coastal acid sulfate soil landscape</article-title>. <source>Soil Res.</source> <volume>52</volume>, <fpage>327</fpage>&#x2013;<lpage>339</lpage>. <pub-id pub-id-type="doi">10.1071/sr13314</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hudd</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2000</year>). <source>Springtime episodic acidification as a regulatory factor of estuary spawing fish recruitment. PhD Thesis</source>. <publisher-loc>Finland</publisher-loc>: <publisher-name>Helsinki University</publisher-name>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jafari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Finkeb</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Van de Wauwb</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ayoubi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Khademi</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Spatial prediction of USDA-great soil groups in the arid zarand region, Iran: Comparing logistic regression approaches to predict diagnostic horizons and soil types</article-title>. <source>Eur. J. Soil Sci.</source> <volume>63</volume>, <fpage>284</fpage>&#x2013;<lpage>298</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-2389.2012.01425.x</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Keskin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Grunwald</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Harris</surname>
<given-names>W. G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Digital mapping of soil carbon fractions with machine learning</article-title>. <source>Geoderma</source> <volume>339</volume>, <fpage>40</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2018.12.037</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kohavi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>John</surname>
<given-names>G. H.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Wrappers for features subset selection</article-title>. <source>Artif. Intell.</source> <volume>97</volume>, <fpage>1</fpage>&#x2013;<lpage>2</lpage>.</citation>
</ref>
<ref id="B48">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Korpela</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Niemel&#xe4;</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>1985</year>). <source>Maaper&#xe4;kartat 1:20 000 ja 1:50 000</source>. <comment>Maank&#xe4;ytt&#xf6; 2</comment>.</citation>
</ref>
<ref id="B49">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kuhn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2013</year>). <source>Applied predictive modeling</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lacoste</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mulder</surname>
<given-names>V. L.</given-names>
</name>
<name>
<surname>Richer-de-Forges</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Martin</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Arrouays</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Evaluating large-extent spatial modeling approaches: A case study for soil depth for France</article-title>. <source>Geoderma Reg.</source> <volume>7</volume>, <fpage>137</fpage>&#x2013;<lpage>152</lpage>. <pub-id pub-id-type="doi">10.1016/j.geodrs.2016.02.006</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lehtinen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Nurmi</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>R&#xe4;m&#xf6;</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1998</year>). <source>Suomen kallioper&#xe4;: 3000 vuosimiljoonaa</source>. <publisher-loc>Helsinki</publisher-loc>: <publisher-name>Suomen Geologinen Seura ry.</publisher-name>, <fpage>375</fpage>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lemercier</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lacoste</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Loum</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Walter</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Extrapolation at regional scale of local soil knowledge using boosted classification trees: A two-step approach</article-title>. <source>Geoderma</source> <volume>171-172</volume>, <fpage>75</fpage>&#x2013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2011.03.010</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Niu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Improving soil thickness estimations based on multiple environmental variables with stacking ensemble methods</article-title>. <source>Remote Sens.</source> <volume>12</volume>, <fpage>3609</fpage>. <pub-id pub-id-type="doi">10.3390/rs12213609</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lie</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Glaser</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huwe</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Uncertainty in the spatial prediction of soil texture: Comparison of regression tree and Random Forest models</article-title>. <source>Geoderma</source> <volume>15</volume>, <fpage>70</fpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2011.10.010</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lie</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Schmidt</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Glaser</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Improving the spatial prediction of soil organic carbon stocks in a complex tropical mountain landscape by methodological specifications in machine learning approaches</article-title>. <source>PLoS ONE</source> <volume>11</volume> (<issue>4</issue>), <fpage>e0153673</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0153673</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>Y.-Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.-G.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>X.-D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G-L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An integrated method of selecting environmental covariates for predictive soil depth mapping</article-title>. <source>J. Integr. Agric.</source> <volume>18</volume> (<issue>2</issue>), <fpage>301</fpage>&#x2013;<lpage>315</lpage>. <pub-id pub-id-type="doi">10.1016/s2095-3119(18)61936-7</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Madsen</surname>
<given-names>H. B.</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>N. H.</given-names>
</name>
</person-group> (<year>1988</year>). <article-title>Potentially acid sulfate soils in relation to landforms and geology</article-title>. <source>Catena</source> <volume>15</volume>, <fpage>137</fpage>&#x2013;<lpage>145</lpage>. <pub-id pub-id-type="doi">10.1016/0341-8162(88)90025-2</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mattb&#xe4;ck</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Boman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>&#xd6;sterholm</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Hydrogeochemical impact of coarse-grained post-glacial acid sulfate soil materials</article-title>. <source>Geoderma</source> <volume>308</volume>, <fpage>291</fpage>&#x2013;<lpage>301</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2017.05.036</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McBratney</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mendon&#xe7;a Santos</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>Minasny</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>On digital soil mapping</article-title>. <source>Geoderma</source> <volume>117</volume>, <fpage>3</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1016/s0016-7061(03)00223-4</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michael</surname>
<given-names>P. S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Ecological impacts and management of acid sulphate soil: A review</article-title>. <source>Asian J. Water, Environ. Pollut.</source> <volume>10</volume> (<issue>No. 4</issue>), <fpage>13</fpage>&#x2013;<lpage>24</lpage>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moradpour</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Entezari</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ayoubi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Karimi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Naimi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Digital exploration of selected heavy metals using Random Forest and a set of environmental covariates at the watershed scale</article-title>. <source>J. Hazard. Mater.</source> <volume>455</volume>, <fpage>131609</fpage>. <pub-id pub-id-type="doi">10.1016/j.jhazmat.2023.131609</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>M&#xfc;ller</surname>
<given-names>A. C.</given-names>
</name>
<name>
<surname>Guido</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <source>An introduction to machine learning with Python</source>. <publisher-loc>Sebastopol, CA 95472</publisher-loc>: <publisher-name>O&#x2019;Reilly Media, Inc., 1005 Gravenstein Highway North</publisher-name>.</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Osl</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dreiseitl</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Cerqueira</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Netzer</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pfeifer</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Demoting redundant features to improve the discriminatory ability in cancer data</article-title>. <source>J. Biomed. Inf.</source> <volume>42</volume>, <fpage>721</fpage>&#x2013;<lpage>725</lpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2009.05.006</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#xd6;sterholm</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>&#xc5;str&#xf6;m</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Spatial trends and losses of major and trace elements in agricultural acid sulphate soils distributed in the artificially drained Rintala area, W. Finland</article-title>. <source>W. Finl. Appl. Geochem. Vol.</source> <volume>17</volume> (<issue>9</issue>), <fpage>1209</fpage>&#x2013;<lpage>1218</lpage>. <pub-id pub-id-type="doi">10.1016/s0883-2927(01)00133-0</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#xd6;sterholm</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>&#xc5;str&#xf6;m</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sundstr&#xf6;m</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Assessment of aquatic pollution, remedial measures and juridical obligations of an acid sulphate soil area in Western Finland</article-title>. <source>Agric. Food Sci.</source> <volume>14</volume>, <fpage>44</fpage>&#x2013;<lpage>56</lpage>. <pub-id pub-id-type="doi">10.2137/1459606054224101</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Palko</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1994</year>). <source>Acid sulphate soils and their agricultural and environmental problems in Finland</source>. <publisher-loc>Finland</publisher-loc>: <publisher-name>Acta University Oulu, C75. University Oulu</publisher-name>.</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Varoquaux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michel</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Thirion</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Grisel</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Scikit-learn: Machine learning in Python</article-title>. <source>J. Mach. Learn. Res.</source> <volume>12</volume>, <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>.</citation>
</ref>
<ref id="B68">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pernu</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1991</year>). <article-title>Model and field studies of direct current resistivity measurements with the combined (half-Schlumberger) array Amn</article-title>. <source>MNB Acta Univ. Ouluensis, Ser. A, Sci. Rerum Nat.</source> <volume>221</volume>, <fpage>123</fpage>.</citation>
</ref>
<ref id="B69">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pons</surname>
<given-names>L. J.</given-names>
</name>
</person-group> (<year>1973</year>). &#x201c;<article-title>Outline of the Genesis,characteristics, classification and improvement of acid sulfate soils</article-title>,&#x201d; in <source>Acid sulphate soils, Introductory papers and bibliography, ILRI Publication 18</source>. <source>Proceedings of the international symposium 13-20</source>. Editor <person-group person-group-type="editor">
<name>
<surname>Dost</surname>
<given-names>H.</given-names>
</name>
</person-group> (<publisher-loc>Wageningen</publisher-loc>), <fpage>3</fpage>&#x2013;<lpage>27</lpage>.</citation>
</ref>
<ref id="B70">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Porwal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Carranza</surname>
<given-names>E. J. M.</given-names>
</name>
<name>
<surname>Hale</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Artificial neural networks for mineral potential mapping: A case study from aravalli province, western India</article-title>. <source>Nat. Resour. Res.</source> <volume>12</volume> (<issue>3</issue>), <fpage>155</fpage>&#x2013;<lpage>171</lpage>. <pub-id pub-id-type="doi">10.1023/a:1025171803637</pub-id>
</citation>
</ref>
<ref id="B71">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Powers</surname>
<given-names>D. M. W.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Evaluation: From precision, recall, and F-measure to ROC, informedness, markedness &#x26; correlation</article-title>. <source>J. Mach. Learn. Technol. V</source> <volume>2</volume>, <fpage>37</fpage>&#x2013;<lpage>63</lpage>.</citation>
</ref>
<ref id="B72">
<citation citation-type="web">
<collab>QGIS Development Team</collab> (<year>2019</year>). <article-title>QGIS geographic information system</article-title>. <comment>Open Source Geospatial Foundation Project <ext-link ext-link-type="uri" xlink:href="http://qgis.osgeo.org">http://qgis.osgeo.org</ext-link>.</comment>
</citation>
</ref>
<ref id="B73">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roos</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>&#xc5;str&#xf6;m</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Gulf of Bothnia receives high concentrations of potentially toxic metals from acid sulphate soils</article-title>. <source>Boreal Environ. Res.</source> <volume>11</volume>, <fpage>383</fpage>&#x2013;<lpage>388</lpage>.</citation>
</ref>
<ref id="B74">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saeys</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Inza</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Larra&#xf1;aga</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>A review of feature selection techniques in bioinformatics</article-title>. <source>Bioinformatics</source> <volume>23</volume> (<issue>Issue 19</issue>), <fpage>2507</fpage>&#x2013;<lpage>2517</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btm344</pub-id>
</citation>
</ref>
<ref id="B75">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmidt</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Behrens</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Daumann</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ramirez-Lopez</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Werban</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Dietrich</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>A comparison of calibration sampling schemes at the field scale</article-title>. <source>Geoderma</source> <volume>232&#x2013;234</volume>, <fpage>243</fpage>&#x2013;<lpage>256</lpage>. <pub-id pub-id-type="doi">10.1016/j.geoderma.2014.05.013</pub-id>
</citation>
</ref>
<ref id="B76">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sindayiheburaa</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ottoyb</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dondeyneb</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Van Meirvennec</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Van Orshovenb</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Comparing digital soil mapping techniques for organic carbon and clay content: Case study in Burundi&#x2019;s central plateaus</article-title>. <source>Catena</source> <volume>156</volume>, <fpage>161</fpage>&#x2013;<lpage>175</lpage>. <pub-id pub-id-type="doi">10.1016/j.catena.2017.04.003</pub-id>
</citation>
</ref>
<ref id="B77">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Teng</surname>
<given-names>H. T.</given-names>
</name>
<name>
<surname>Viscarra Rossel</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Behrens</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Updating a national soil classification with spectroscopic predictions and digital soil mapping</article-title>. <source>Catena</source> <volume>164</volume>, <fpage>125</fpage>&#x2013;<lpage>134</lpage>. <pub-id pub-id-type="doi">10.1016/j.catena.2018.01.015</pub-id>
</citation>
</ref>
<ref id="B78">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tesfa</surname>
<given-names>T. K.</given-names>
</name>
<name>
<surname>Tarboton</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Chandler</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>McNamara</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Modeling soil depth from topographic and land cover attributes</article-title>. <source>Water Resour. Res.</source> <volume>45</volume>, <fpage>1</fpage>&#x2013;<lpage>16</lpage>. <pub-id pub-id-type="doi">10.1029/2008wr007474</pub-id>
</citation>
</ref>
<ref id="B79">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tziachrisa</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Aschonitisa</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Chatzistathisa</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Papadopoulou</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Assessment of spatial hybrid methods for predicting soil organic matter using DEM derivatives and soil parameters</article-title>. <source>Catena</source> <volume>174</volume>, <fpage>206</fpage>&#x2013;<lpage>216</lpage>. <pub-id pub-id-type="doi">10.1016/j.catena.2018.11.010</pub-id>
</citation>
</ref>
<ref id="B80">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Urho</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2002</year>). <source>The importance of larvae and nursery areas for fish production</source>. <publisher-loc>Finland</publisher-loc>: <publisher-name>Helsinki University</publisher-name>, <fpage>135</fpage>.</citation>
</ref>
<ref id="B81">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Van Rossum</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Drake</surname>
<given-names>F. L.</given-names>
</name>
</person-group> (<year>2009</year>). <source>Python 3 reference manual, scotts valley</source>. <publisher-loc>CA</publisher-loc>: <publisher-name>CreateSpace</publisher-name>.</citation>
</ref>
<ref id="B82">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Veronesi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Schillaci</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Comparison between geostatistical and machine learning models as predictors of topsoil organic carbon with a focus on local uncertainty estimation</article-title>. <source>Ecol. Indic. V.</source> <volume>101</volume>, <fpage>1032</fpage>&#x2013;<lpage>1044</lpage>. <pub-id pub-id-type="doi">10.1016/j.ecolind.2019.02.026</pub-id>
</citation>
</ref>
<ref id="B83">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Dunbrack</surname>
<given-names>R. L.</given-names>
<suffix>Jr.</suffix>
</name>
</person-group> (<year>2013</year>). <article-title>The role of balanced training and testing data sets for binary classifiers in bioinformatics</article-title>. <source>PLOS ONE</source> <volume>8</volume> (<issue>7</issue>), <fpage>e67863</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0067863</pub-id>
</citation>
</ref>
<ref id="B84">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weiss</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Provost</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>The effect of class distribution on classifier learning: An empirical study</article-title>. <source>Tech. Rep</source>.</citation>
</ref>
<ref id="B85">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wiesmeier</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Barthold</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Blank</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>K&#xf6;gel-Knabner</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Digital mapping of soil organic matter stocks using Random Forest modeling in a semi-arid steppe ecosystem</article-title>. <source>Plant Soil</source> <volume>340</volume>, <fpage>7</fpage>&#x2013;<lpage>24</lpage>. <pub-id pub-id-type="doi">10.1007/s11104-010-0425-z</pub-id>
</citation>
</ref>
<ref id="B86">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Grunwald</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Myers</surname>
<given-names>D. B.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Harris</surname>
<given-names>W. G.</given-names>
</name>
<name>
<surname>Comerford</surname>
<given-names>N. B.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Holistic environmental soil-landscape modeling of soil organic carbon</article-title>.<source>/ Environ. Model. Softw.</source> <volume>57</volume>, <fpage>202</fpage>&#x2013;<lpage>215</lpage>. <pub-id pub-id-type="doi">10.1016/j.envsoft.2014.03.004</pub-id>
</citation>
</ref>
<ref id="B87">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yli-Halla</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mokma</surname>
<given-names>D. L.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Soil temperature regimes in Finland</article-title>. <source>Agric. food Sci. Finl.</source> <volume>7</volume>, <fpage>507</fpage>&#x2013;<lpage>512</lpage>. <pub-id pub-id-type="doi">10.23986/afsci.5606</pub-id>
</citation>
</ref>
<ref id="B88">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yli-Halla</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Puustinen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Koskiaho</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1999</year>). <article-title>Area of cultivated acid sulfate soils in Finland</article-title>. <source>Soil Use Manag.</source> <volume>15</volume>, <fpage>62</fpage>&#x2013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1111/j.1475-2743.1999.tb00065.x</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>