<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Earth Sci.</journal-id>
<journal-title>Frontiers in Earth Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Earth Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-6463</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1267386</article-id>
<article-id pub-id-type="doi">10.3389/feart.2023.1267386</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Earth Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Machine-learning models to predict P- and S-wave velocity profiles for Japan as an example</article-title>
<alt-title alt-title-type="left-running-head">Kim et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/feart.2023.1267386">10.3389/feart.2023.1267386</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Kim</surname>
<given-names>Jisong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2377497/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Kang</surname>
<given-names>Jae-Do</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2429482/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Kim</surname>
<given-names>Byungmin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2052933/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Civil, Urban, Earth, and Environmental Engineering</institution>, <institution>Ulsan National Institute of Science and Technology</institution>, <addr-line>Ulsan</addr-line>, <country>Republic of Korea</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Earthquake Disaster Mitigation Center</institution>, <institution>Seoul Institute of Technology</institution>, <addr-line>Seoul</addr-line>, <country>Republic of Korea</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/88099/overview">Biswajeet Pradhan</ext-link>, University of Technology Sydney, Australia</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1623541/overview">Pijush Samui</ext-link>, National Institute of Technology Patna, India</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1952949/overview">Xianyang Qi</ext-link>u, Central South University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Byungmin Kim, <email>byungmin.kim@unist.ac.kr</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1267386</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>09</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Kim, Kang and Kim.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Kim, Kang and Kim</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Wave velocity profiles are significant for various fields, including rock engineering, petroleum engineering, and earthquake engineering. However, direct measurements of wave velocities are often constrained by time, cost, and site conditions. If wave velocity measurements are unavailable, they need to be estimated based on other known proxies. This paper proposes machine learning (ML) approaches to predict the compression and shear wave velocities (<italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub>, respectively) in Japan. We utilize borehole databases from two seismograph networks of Japan: Kyoshin Network (K-NET) and Kiban Kyoshin Network (KiK-net). We consider various factors such as depth, N-value, density, slope angle, elevation, geology, soil/rock type, and site coordinates. We use three ML techniques: Gradient Boosting (GB), Random Forest (RF), and Artificial Neural Network (ANN) to develop predictive models for both <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> and evaluate the performances of the models based on root mean squared errors and the five-fold cross-validation method. The GB-based model provides the best estimation of <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> for both seismograph networks. Among the considered factors, the depth, standard penetration test (SPT) N-value, and density have the strongest influence on the wave velocity estimation for K-NET. For KiK-net, the depth and site longitude have the strongest influence. The study confirms the applicability of commonly used machine-learning techniques in predicting wave velocities, and implies that exploring additional factors will enhance the performance.</p>
</abstract>
<kwd-group>
<kwd>shear wave velocity</kwd>
<kwd>compression wave velocity</kwd>
<kwd>machine learning</kwd>
<kwd>gradient boosting</kwd>
<kwd>random forest</kwd>
<kwd>artificial neural network</kwd>
<kwd>cross-validation</kwd>
</kwd-group>
<contract-sponsor id="cn001">Korea Hydro and Nuclear Power<named-content content-type="fundref-id">10.13039/501100018792</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Ministry of Land, Infrastructure and Transport<named-content content-type="fundref-id">10.13039/501100003565</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Geoinformatics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Compression and shear wave velocities (<italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub>, respectively) are often employed to assess the properties of underground rock environments and to design geotechnical projects. <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> are significant in various fields such as rock mechanical property calculations (<xref ref-type="bibr" rid="B12">Chang et al., 2006</xref>; <xref ref-type="bibr" rid="B3">Ameen et al., 2009</xref>; <xref ref-type="bibr" rid="B28">Jamshidi et al., 2018</xref>; <xref ref-type="bibr" rid="B47">Rahman and Sarkar, 2021</xref>), pore structure identification (<xref ref-type="bibr" rid="B17">Eberli et al., 2003</xref>; <xref ref-type="bibr" rid="B40">Panza et al., 2019</xref>), lithology determination (<xref ref-type="bibr" rid="B45">Pickett, 1963</xref>; <xref ref-type="bibr" rid="B14">Deng et al., 2017</xref>), fluid saturation (<xref ref-type="bibr" rid="B51">Si et al., 2016</xref>; <xref ref-type="bibr" rid="B48">Roy et al., 2017</xref>; <xref ref-type="bibr" rid="B15">Ding et al., 2019</xref>), seismic liquefaction (<xref ref-type="bibr" rid="B49">Samui et al., 2011</xref>; <xref ref-type="bibr" rid="B31">Karthikeyan and Samui, 2014</xref>; <xref ref-type="bibr" rid="B29">Jena et al., 2023</xref>), seismic site responses, and ground motion predictions (<xref ref-type="bibr" rid="B18">Fiorentino et al., 2019</xref>; <xref ref-type="bibr" rid="B24">Harmon et al., 2019</xref>; <xref ref-type="bibr" rid="B32">Kim, 2019</xref>). Such wave velocities are measured by invasive tests such as down-hole test, cross-hole test, and suspension PS logging, as well as non-destructive tests such as Multichannel Analysis of Surface Wave (MASW), Spectral Analysis of Surface Wave (SASW), and Multichannel Simulation with One Receiver (MSOR). However, these tests are often constrained by time, cost, and site conditions (<xref ref-type="bibr" rid="B25">Hasancebi and Ulusay, 2007</xref>; <xref ref-type="bibr" rid="B5">Anemangely et al., 2019</xref>; <xref ref-type="bibr" rid="B58">Xiao et al., 2021</xref>).</p>
<p>To address the problems mentioned in the prior paragraph, numerous researchers have proposed indirect methods to estimate <italic>V</italic>
<sub>
<italic>P</italic>
</sub> or <italic>V</italic>
<sub>
<italic>S</italic>
</sub>. For instance, several studies have presented the relationships between <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and the mechanical properties of rock materials, such as uniaxial compressive strength (<xref ref-type="bibr" rid="B41">Pappalardo, 2015</xref>), density (<xref ref-type="bibr" rid="B59">Yasar and Erdogan, 2004</xref>), and porosity (<xref ref-type="bibr" rid="B54">Sousa et al., 2005</xref>). Various correlation models between <italic>V</italic>
<sub>
<italic>S</italic>
</sub> and standard penetration test (SPT) resistance (N-value) have also been suggested (e.g., <xref ref-type="bibr" rid="B39">Ohta and Goto, 1978</xref>; <xref ref-type="bibr" rid="B4">Andrus et al., 2004</xref>; <xref ref-type="bibr" rid="B2">Akin et al., 2011</xref>; <xref ref-type="bibr" rid="B52">Sil and Haloi, 2017</xref>; <xref ref-type="bibr" rid="B7">Bajaj and Anbazhagan, 2019</xref>). For example, <xref ref-type="bibr" rid="B36">Kwak et al. (2015)</xref>; <xref ref-type="bibr" rid="B56">Tsai et al. (2019)</xref> inferred <italic>V</italic>
<sub>
<italic>S</italic>
</sub> using empirical equations conditioned on the N-value and other independent variables such as vertical effective stress and soil type. <xref ref-type="bibr" rid="B46">Rahimi et al. (2020)</xref> presented the effect of soil aging on SPT-<italic>V</italic>
<sub>
<italic>S</italic>
</sub> correlations. Furthermore, researchers have predicted the time-averaged shear-wave velocity in the upper 30&#xa0;m of soil deposits (<italic>V</italic>
<sub>
<italic>S</italic>30</sub>) based on various proxies, such as topographic slope, surface geology, elevation, and terrain type (e.g., <xref ref-type="bibr" rid="B34">Kottke et al., 2012</xref>; <xref ref-type="bibr" rid="B42">Parker et al., 2017</xref>; <xref ref-type="bibr" rid="B37">Kwok et al., 2018</xref>; <xref ref-type="bibr" rid="B26">Heath et al., 2020</xref>).</p>
<p>The demand for Machine learning (ML) applications has been increasing as huge volumes of data are accessible over a computer network. ML algorithms are well suited for making regression models on complex data-driven problems. Researchers have studied <italic>V</italic>
<sub>
<italic>P</italic>
</sub> or <italic>V</italic>
<sub>
<italic>S</italic>
</sub> estimation based on ML (e.g., <xref ref-type="bibr" rid="B53">Singh and Kanli, 2016</xref>; <xref ref-type="bibr" rid="B43">Paul et al., 2018</xref>; <xref ref-type="bibr" rid="B5">Anemangely et al., 2019</xref>; <xref ref-type="bibr" rid="B16">Dumke and Berndt, 2019</xref>; <xref ref-type="bibr" rid="B57">Wang and Peng, 2019</xref>; <xref ref-type="bibr" rid="B61">Zhang et al., 2020</xref>). In particular, <xref ref-type="bibr" rid="B16">Dumke and Berndt (2019)</xref> used the Random Forest (RF) regression algorithm to estimate <italic>V</italic>
<sub>
<italic>P</italic>
</sub> as a function of depth on global marine locations. They used data from 333 boreholes and considered 38 geological variables, such as site coordinates, sediment thickness, and depth below the seafloor. They validate the ML model using 10-fold cross-validation (CV). <xref ref-type="bibr" rid="B43">Paul et al. (2018)</xref> used an Artificial Neural Network (ANN) algorithm on data from five wells in India to estimate the <italic>V</italic>
<sub>
<italic>P</italic>
</sub>. <xref ref-type="bibr" rid="B53">Singh and Kanli (2016)</xref> used an ANN to estimate <italic>V</italic>
<sub>
<italic>S</italic>
</sub> in an oil field located in southeastern Turkey. <xref ref-type="bibr" rid="B5">Anemangely et al. (2019)</xref> adopted the least square version of the support vector machine (LSSVM) algorithm combined with three optimization algorithms to predict <italic>V</italic>
<sub>
<italic>S</italic>
</sub> using data from two oilfields located in the southwest of Iran.</p>
<p>This study aims to train the three&#xa0;ML algorithms (i.e., gradient boosting, random forest, and artificial neural network) to estimate both <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> in Japan. We utilize borehole databases, covering all of Japan from two seismograph networks: Kyoshin Network (K-NET) and Kiban Kyoshin Network (KiK-net). We consider various factors such as depth, N-value, density, slope angle, elevation, geology, soil/rock type, and site coordinates. We quantitatively evaluate the prediction performances of the ML-based algorithms based on five-fold cross-validation and evaluate the relative importance of the factors.</p>
</sec>
<sec id="s2">
<title>2 Data</title>
<p>In this study, we obtained site data from two seismograph networks of Japan, Kyoshin Network (K-NET) and Kiban Kyoshin Network (KiK-net), where the <xref ref-type="bibr" rid="B38">National Research Institute for Earth Science and Disaster Resilience (2019)</xref> has operated since 1996. Each site of these two networks has profiles of <italic>V</italic>
<sub>
<italic>P</italic>
</sub>, <italic>V</italic>
<sub>
<italic>S</italic>
</sub>, and soil/rock types. In addition, the K-NET site has profiles of standard penetration test (SPT) resistance values (N-values) and density with a depth interval of 1&#xa0;m. The energy efficiency is unknown for the borings at the K-NET sites (<xref ref-type="bibr" rid="B36">Kwak et al., 2015</xref>). Therefore, we utilized unnormalized N-values. Because of the inconsistent datasets between the two seismograph networks, we considered training the ML models for each network.</p>
<p>For the datasets, the velocity profile data were resampled to a depth interval of 1&#xa0;m. For all of the K-NET sites, a minimum depth interval is 1&#xa0;m. Furthermore, approximately 43% of KiK-net sites have minimum depth intervals of 1&#xa0;m or shorter. Therefore, we consider that resampling the profile data into a depth interval of 1&#xa0;m is reasonable. We also screen the suspicious profile data such as those with the velocity of zero. In addition to the depth-dependent variables provided by the networks, we also considered the following five depth-independent variables: site latitude, site longitude, geology, topographic slope angle, and elevation. The geology map was obtained from the seamless digital geological map of Japan (1:200,000) (<xref ref-type="bibr" rid="B21">Geological Survey of Japan, 2015</xref>), and the slope angle and elevation were obtained from the digital elevation map (DEM) of the Shuttle Radar Topography Mission (SRTM) with a resolution of 30&#xa0;m. We then used the nine independent variables (i.e., site longitude, site latitude, geology, slope angle, elevation, N-value, density, depth, soil/rock type) for the K-NET, and seven (i.e., site longitude, site latitude, geology, slope angle, elevation, depth, and soil/rock type) for the KiK-net sites, as summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Datasets used in this study.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Seismograph network</th>
<th align="center">Dependent variables</th>
<th align="center">Independent variables</th>
<th align="center">Description</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="9" align="center">K-NET</td>
<td align="center">
<italic>V</italic>
<sub>
<italic>P</italic>
</sub>
</td>
<td align="left">Site longitude</td>
<td align="left">1) 996 sites (<italic>V</italic>
<sub>
<italic>P</italic>
</sub>)</td>
</tr>
<tr>
<td rowspan="8" align="center">
<italic>V</italic>
<sub>
<italic>S</italic>
</sub>
</td>
<td align="left">Site latitude</td>
<td align="left">2) 996 sites (<italic>V</italic>
<sub>
<italic>S</italic>
</sub>)</td>
</tr>
<tr>
<td align="left">Geology</td>
<td align="left">3) 15,253 data samples (<italic>V</italic>
<sub>
<italic>P</italic>
</sub>)</td>
</tr>
<tr>
<td align="left">Slope angle</td>
<td align="left">4) 15,253 data samples (<italic>V</italic>
<sub>
<italic>S</italic>
</sub>)</td>
</tr>
<tr>
<td align="left">Elevation</td>
<td rowspan="5" align="left">5) Velocity profiles were sampled at every 1-m depth interval</td>
</tr>
<tr>
<td align="left">N-value</td>
</tr>
<tr>
<td align="left">Density</td>
</tr>
<tr>
<td align="left">Depth</td>
</tr>
<tr>
<td align="left">Soil/rock type</td>
</tr>
<tr>
<td rowspan="7" align="center">KiK-net</td>
<td align="center">
<italic>V</italic>
<sub>
<italic>P</italic>
</sub>
</td>
<td align="left">Site longitude</td>
<td align="left">1) 677 sites (<italic>V</italic>
<sub>
<italic>P</italic>
</sub>)</td>
</tr>
<tr>
<td rowspan="8" align="center">
<italic>V</italic>
<sub>
<italic>S</italic>
</sub>
</td>
<td align="left">Site latitude</td>
<td align="left">2) 675 sites (<italic>V</italic>
<sub>
<italic>S</italic>
</sub>)</td>
</tr>
<tr>
<td align="left">Geology</td>
<td align="left">3) 136,315 data samples (<italic>V</italic>
<sub>
<italic>P</italic>
</sub>)</td>
</tr>
<tr>
<td align="left">Slope angle</td>
<td align="left">4) 132,855 data samples (<italic>V</italic>
<sub>
<italic>S</italic>
</sub>)</td>
</tr>
<tr>
<td align="left">Elevation</td>
<td rowspan="3" align="left">5) Velocity profiles were sampled at every 1-m depth interval</td>
</tr>
<tr>
<td align="left">Depth</td>
</tr>
<tr>
<td align="left">Soil/rock type</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We considered all sites where all of the variables were available: 996&#xa0;K-NET sites with 15,253 data samples for each of <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> and 677 KiK-net sites with 136,315 data samples for <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and 132,855 data samples for <italic>V</italic>
<sub>
<italic>S</italic>
</sub>. The dataset information is summarized in <xref ref-type="table" rid="T1">Table 1</xref>. The considered sites (i.e., recording stations) covering Japan are shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Locations of K-NET and KiK-net recording sites used in this study.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g001.tif"/>
</fig>
<p>The distributions of the numerical variables for the K-NET and KiK-net datasets are shown in <xref ref-type="fig" rid="F2">Figure 2</xref> and <xref ref-type="fig" rid="F3">Figure 3</xref>, respectively. The depth to the bottom of the borehole (D<sup>bh</sup>) at the K-NET sites ranges from 5 to 20&#xa0;m, with 83% concentration at 10&#xa0;m and 20&#xa0;m (<xref ref-type="fig" rid="F2">Figure 2A</xref>). The elevation ranges from &#x2212;3&#xa0;m to 1,502&#xa0;m, 75% of which are positioned under 179&#xa0;m, as shown in the boxplot above the histogram (<xref ref-type="fig" rid="F2">Figure 2B</xref>). The slope angle ranges from 0&#xb0; to 30.87&#xb0;, with 75% below 5.26&#xb0; (<xref ref-type="fig" rid="F2">Figure 2C</xref>). The 96 outliers are observed as circular forms in each boxplot (<xref ref-type="fig" rid="F2">Figures 2B, C</xref>). The N-value with depth ranges from 0 to 500 with 69% below 90, where the four outliers are observed: three of which are 375, and one is 500 (<xref ref-type="fig" rid="F2">Figure 2D</xref>). The density with depth is distributed from 0.69&#xa0;g/cm<sup>3</sup> to 2.82&#xa0;g/cm<sup>3</sup> with 75% under 1.98&#xa0;g/cm<sup>3</sup>, in which 306 outliers are detected (<xref ref-type="fig" rid="F2">Figure 2E</xref>). The <italic>V</italic>
<sub>
<italic>S</italic>
</sub> is distributed from 37&#xa0;m/s to 2,350&#xa0;m/s with 75% slower than 450&#xa0;m/s (<xref ref-type="fig" rid="F2">Figure 2F</xref>). The <italic>V</italic>
<sub>
<italic>P</italic>
</sub> ranges from 140&#xa0;m/s to 5,270&#xa0;m/s with 75% slower than 1,800&#xa0;m/s (<xref ref-type="fig" rid="F2">Figure 2G</xref>). For categorical variables in K-NET sites in our dataset, 12 unique soil/rock types according to depth and 110 unique types of geology are observed.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Variables distribution for K-NET dataset used in this study: <bold>(A)</bold> depth to the bottom of the borehole (D<sup>bh</sup>), <bold>(B)</bold> elevation, <bold>(C)</bold> slope angle for the sites (i.e., 996 sites), and <bold>(D)</bold> N-value, <bold>(E)</bold> density, <bold>(F)</bold> <italic>V</italic>
<sub>
<italic>S</italic>
</sub>, <bold>(G)</bold> <italic>V</italic>
<sub>
<italic>P</italic>
</sub> for data samples (i.e., 15,253 data samples). In the boxplot above the histogram, the blue line represents a median value, and the box represents 25 and 75 percentiles of the data.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g002.tif"/>
</fig>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Variables distribution for KiK-net dataset used in this study: <bold>(A)</bold> depth to the bottom of the borehole (D<sup>bh</sup>), <bold>(B)</bold> station elevation, <bold>(C)</bold> station slope angle for the sites of <italic>V</italic>
<sub>
<italic>P</italic>
</sub> dataset (i.e., 677 sites), and <bold>(D)</bold> <italic>V</italic>
<sub>
<italic>S</italic>
</sub>, <bold>(E)</bold> <italic>V</italic>
<sub>
<italic>P</italic>
</sub> for data samples (i.e., 132,855 and 136,315 data samples, respectively). In the boxplot above the histogram, the blue line represents a median value, and the box represents 25 and 75 percentiles of the data.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g003.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F3">Figure 3A</xref> depicts the D<sup>bh</sup> of the KiK-net sites ranging from 92&#xa0;m to 2,000 m, with 75% under 199&#xa0;m, where 46 outliers are observed. <xref ref-type="fig" rid="F3">Figure 3B</xref> presents the site elevation ranging from &#x2212;5 m to 1,302&#xa0;m with 75% below 330&#xa0;m, where 34 outliers are detected. <xref ref-type="fig" rid="F3">Figure 3C</xref> shows the slope angle that ranges from 0&#xb0; to 36.23&#xb0; with 75% under 10.42&#xb0;, where 12 outliers are observed. <xref ref-type="fig" rid="F3">Figure 3D</xref> shows the <italic>V</italic>
<sub>
<italic>S</italic>
</sub> ranging from 20&#xa0;m/s to 3,500&#xa0;m/s with 75% slower than 1,720&#xa0;m/s. <xref ref-type="fig" rid="F3">Figure 3E</xref> presents the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> that ranges from 50&#xa0;m/s to 6,100&#xa0;m/s with 75% slower than 3,830&#xa0;m/s. For categorical variables, soil/rock type with depth has 588 unique features, among which the features include the various combinations of several soil types, such as &#x2018;sand and gravel&#x2019;, &#x2018;shale with gravel&#x2019;, and &#x2018;sandstone and mudstone&#x2019;. Furthermore, 119 unique geological classes are observed.</p>
</sec>
<sec id="s3">
<title>3 Machine learning (ML) models</title>
<p>The ML model uses the following variables: the depth and depth-related information (i.e., N-value, density, soil/rock type), and site information (i.e., coordinates, slope angle, elevation, geology) described in <xref ref-type="table" rid="T1">Table 1</xref> to infer <italic>V</italic>
<sub>
<italic>P</italic>
</sub> or <italic>V</italic>
<sub>
<italic>S</italic>
</sub> on a specific depth (e.g., 15&#xa0;m) of the site. This section describes the ML algorithms utilized for <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> prediction. We illustrated them using all K-NET data samples for <italic>V</italic>
<sub>
<italic>S</italic>
</sub> as an example. We used Scikit-learn (<xref ref-type="bibr" rid="B44">Pedregosa et al., 2011</xref>) for the implementation of Gradient Boosting (GB) and Random Forest (RF) algorithms and the Tensorflow (<xref ref-type="bibr" rid="B1">Abadi et al., 2016</xref>) for the Artificial Neural Network (ANN) algorithm. Note that comparing these three algorithms is a popular practice in the field of machine learning-based studies (e.g., <xref ref-type="bibr" rid="B35">Krauss et al., 2017</xref>; <xref ref-type="bibr" rid="B33">Kim et al., 2020</xref>; <xref ref-type="bibr" rid="B30">Jun, 2021</xref>; <xref ref-type="bibr" rid="B50">Seo et al., 2022</xref>). These methods represent different types of machine learning algorithms and have been proven effective in handling complicated relationships within various datasets. Given their proven reliability, we employed such methods to assess the generalization performance in predicting velocities on the dataset utilized in this study. Furthermore, the hyperparameters used in this study were taken from the suggestions mentioned in the following subsections to present the results of baseline solutions, serving as a fundamental benchmark for assessing their effectiveness in predicting velocities.</p>
<sec id="s3-1">
<title>3.1 Gradient boosting (GB)</title>
<p>Before we start explaining the GB, we describe the decision tree algorithm, which is the main concept of GB and RF. The decision tree consists of nodes, where a tree is grown on the training dataset. The tree contains three types of nodes: root node, internal node, and leaf node, where the root and internal nodes play a role in splitting the data samples, and the leaf node makes the final decision for the prediction value.</p>
<p>We presented an example tree using the independent variables of the K-NET, as shown in <xref ref-type="fig" rid="F4">Figure 4</xref> to explain the internal structure. First, the root node splits 15,253 independent data samples into two internal nodes by asking if the N-value <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mo>&#x2264;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 45.5. If the condition is true, the internal node condition (i.e., N-value <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mo>&#x2264;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 14.75) works to further divide the allocated data samples into a leaf node (<italic>R</italic>
<sub>
<italic>1</italic>
</sub>) and another internal node. If the root node condition is false, the data samples are further divided by the internal node (i.e., depth <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mo>&#x2264;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 4.5&#xa0;m) into a leaf node (<italic>R</italic>
<sub>
<italic>4</italic>
</sub>) and another internal node. Following the if-else rules, the model finally creates a tree that consists of a root node, four internal nodes, and six leaf nodes (<italic>R</italic>
<sub>
<italic>1</italic>
</sub>, <italic>R</italic>
<sub>
<italic>2</italic>
</sub>, . . ., <italic>R</italic>
<sub>
<italic>6</italic>
</sub>).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Example of the decision tree using the variables of the K-NET dataset.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g004.tif"/>
</fig>
<p>One may wonder how the decision tree model creates the splitting criterion of the node. The model grows a tree by splitting the data samples into two groups by finding the threshold that minimizes the mean of squared errors (MSE), which is calculated as<disp-formula id="e1">
<mml:math id="m4">
<mml:mrow>
<mml:mtext>MSE</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>J</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msubsup>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the measured <italic>V</italic>
<sub>
<italic>S</italic>
</sub> associated with <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> data sample of the K-NET (i.e., <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 1, 2, . . ., <italic>n</italic>; <italic>n</italic> &#x3d; 15,253), and <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msubsup>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the estimated <italic>V</italic>
<sub>
<italic>S</italic>
</sub> determined by a specific leaf node, <italic>R</italic>
<sub>
<italic>j</italic>
</sub>, where <italic>j</italic> is the leaf node index (<italic>j</italic> &#x3d; 1, 2, . . ., <italic>J</italic>).</p>
<p>The estimated <italic>V</italic>
<sub>
<italic>S</italic>
</sub> (<inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>), which is an output of the trained tree model (<italic>T</italic>), can be expressed by the following equation:<disp-formula id="e2">
<mml:math id="m10">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>J</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf9">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the independent variables of the <inline-formula id="inf10">
<mml:math id="m12">
<mml:mrow>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> data sample of the K-NET, <inline-formula id="inf11">
<mml:math id="m13">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf12">
<mml:math id="m14">
<mml:mrow>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are the specific leaf node number and the total number of leaf nodes (i.e., six leaf nodes for the example tree in <xref ref-type="fig" rid="F4">Figure 4</xref>), respectively, <inline-formula id="inf13">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted dependent variable decided by the specific region <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is an indicator function that takes a value of 0 or 1 (i.e., <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and 0 otherwise).</p>
<p>However, a single decision tree model is prone to overfitting on a training dataset, resulting in a high variance in new data samples (<xref ref-type="bibr" rid="B22">Geurts et al., 2009</xref>; <xref ref-type="bibr" rid="B13">Czajkowski and Kretowski, 2019</xref>). The GB algorithm, proposed by <xref ref-type="bibr" rid="B19">Friedman (2001)</xref>; <xref ref-type="bibr" rid="B20">Friedman (2002)</xref>, is an ensemble of weak models (i.e., decision trees) and provides robust model performance over the overfitting problem. GB grows many decision trees and connects them in order like links in a chain, where each new tree is grown to modify a mistake made by a previous tree. An example of a GB architecture is presented in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Architecture of ensemble learning methods (Random Forest and Gradient Boosting).</p>
</caption>
<graphic xlink:href="feart-11-1267386-g005.tif"/>
</fig>
<p>The trees in the GB estimate the residuals between <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> instead of the <italic>V</italic>
<sub>
<italic>S</italic>
</sub> itself and are trained to minimize the residuals. The specific steps for training the GB model are as follows. Step 1: The GB has a constant value (<inline-formula id="inf20">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>), which is <inline-formula id="inf21">
<mml:math id="m23">
<mml:mrow>
<mml:mover accent="true">
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> from the training dataset. Step 2: From <inline-formula id="inf22">
<mml:math id="m24">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf23">
<mml:math id="m25">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf24">
<mml:math id="m26">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the last tree index, the GB repeats the following steps (3&#x2013;5) for successive trees (<inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>). Step 3: An individual tree <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> calculates the residual data for each data sample as:<disp-formula id="e3">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf27">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a residual associated with data sample <inline-formula id="inf28">
<mml:math id="m31">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the dataset (i.e., <inline-formula id="inf29">
<mml:math id="m32">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>15,253</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>) and tree index <inline-formula id="inf30">
<mml:math id="m33">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf31">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the prediction value of the previous GB model (<inline-formula id="inf32">
<mml:math id="m35">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>). Step 4: After the residual dataset (<inline-formula id="inf33">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mn>15253</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) for <inline-formula id="inf34">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is developed, <inline-formula id="inf35">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is trained on the dataset, <inline-formula id="inf36">
<mml:math id="m39">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>15253</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, instead of <inline-formula id="inf37">
<mml:math id="m40">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>15253</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. The leaf node (<inline-formula id="inf38">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) is determined during training, where <inline-formula id="inf39">
<mml:math id="m42">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the leaf node index of the tree, <inline-formula id="inf40">
<mml:math id="m43">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The mean residual value predicted in <inline-formula id="inf41">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is <inline-formula id="inf42">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> The <inline-formula id="inf43">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is subsequently reduced by a learning rate (<inline-formula id="inf44">
<mml:math id="m47">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), which is a constant value to reduce the contribution of each tree. Therefore, the tree (<inline-formula id="inf45">
<mml:math id="m48">
<mml:mrow>
<mml:mfenced open="" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> can be described as follows:<disp-formula id="e4">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf46">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mi>J</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of leaf nodes in <inline-formula id="inf47">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The equation returns the <inline-formula id="inf48">
<mml:math id="m52">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mo>&#x2a;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> according to independent data (<inline-formula id="inf49">
<mml:math id="m53">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>). Step 5: After a tree is built, it is added to the previous tree. The updated GB model (<inline-formula id="inf50">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>) can be described as follows:<disp-formula id="e5">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>After the GB finishes developing the last tree (<inline-formula id="inf51">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>), we finally obtain the <inline-formula id="inf52">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which is the complete GB model ready to use for predicting <italic>V</italic>
<sub>
<italic>S</italic>
</sub> (<inline-formula id="inf53">
<mml:math id="m58">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>). In this study, we set the number of trees (<inline-formula id="inf54">
<mml:math id="m59">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) to 100 and the learning rate (<inline-formula id="inf55">
<mml:math id="m60">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) to 0.1, as suggested by <xref ref-type="bibr" rid="B44">Pedregosa et al. (2011)</xref>.</p>
</sec>
<sec id="s3-2">
<title>3.2 Random forest (RF)</title>
<p>The RF algorithm, proposed by <xref ref-type="bibr" rid="B11">Breiman (2001)</xref>, is a bootstrap aggregation (bagging) ensemble algorithm that grows many decision trees using a random subset of the data. Unlike GB, RF trains many weak trees in a parallel manner, where the trees are not affected by each other while being trained. Each tree in the GB predicts the residual value, but the tree in the RF directly returns <inline-formula id="inf56">
<mml:math id="m61">
<mml:mrow>
<mml:msup>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. An example of an RF architecture is presented in <xref ref-type="fig" rid="F5">Figure 5</xref>. From <inline-formula id="inf57">
<mml:math id="m62">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf58">
<mml:math id="m63">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, each tree (<inline-formula id="inf59">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) is grown on bootstrap samples (<inline-formula id="inf60">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), where <inline-formula id="inf61">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the randomly sampled subset data from the training dataset.</p>
<p>While <inline-formula id="inf62">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is trained on <inline-formula id="inf63">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the number of variables (<inline-formula id="inf64">
<mml:math id="m69">
<mml:mrow>
<mml:mfenced open="" close=")" separators="|">
<mml:mrow>
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> in <inline-formula id="inf65">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is randomly chosen from the total number of independent variables (<inline-formula id="inf66">
<mml:math id="m71">
<mml:mrow>
<mml:mfenced open="" close=")" separators="|">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> to minimize the MSE. The node is further split into two child nodes after choosing the best split points among <inline-formula id="inf67">
<mml:math id="m72">
<mml:mrow>
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> number of variables.</p>
<p>After RF completes training all trees, it makes a final decision of <inline-formula id="inf68">
<mml:math id="m73">
<mml:mrow>
<mml:msup>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> at a new data sample <inline-formula id="inf69">
<mml:math id="m74">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> by averaging the multiple results of <inline-formula id="inf70">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which can be described as<disp-formula id="e6">
<mml:math id="m76">
<mml:mrow>
<mml:msup>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>B</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>In this study, we set the number of trees (<inline-formula id="inf71">
<mml:math id="m77">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) to 100 and <inline-formula id="inf72">
<mml:math id="m78">
<mml:mrow>
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (i.e., all variables are considered), as suggested by <xref ref-type="bibr" rid="B44">Pedregosa et al. (2011)</xref>.</p>
</sec>
<sec id="s3-3">
<title>3.3 Artificial neural network (ANN)</title>
<p>The ANN model comprises a collection of nodes grouped in layers, where each node in a layer is connected to the nodes in the next layer. The ANN model includes three types of layers: input layer, hidden layer, and output layer. <xref ref-type="fig" rid="F6">Figure 6</xref> presents an ANN model containing the two hidden layers used in this study. The number of input variables is nine for K-NET and seven for KiK-net, as described in <xref ref-type="table" rid="T1">Table 1</xref>. However, we applied the binary encoding method to categorical variables. The total number of variables was increased to 18 for the K-NET and 22 for the KiK-net to train ML models (i.e., GB, RF, and ANN). A detailed explanation of this is provided in the subsequent section. Therefore, the number of input nodes is 18 for K-NET and 22 for KiK-net. We set the number of nodes in the hidden layers to 200, as inspired by <xref ref-type="bibr" rid="B33">Kim et al. (2020)</xref>. In <xref ref-type="fig" rid="F6">Figure 6</xref>, the values of each node for hidden layer 1 (<inline-formula id="inf73">
<mml:math id="m79">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>), hidden layer 2 (<inline-formula id="inf74">
<mml:math id="m80">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>), and output layer (<inline-formula id="inf75">
<mml:math id="m81">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) can be described as<disp-formula id="e7">
<mml:math id="m82">
<mml:mrow>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>18</mml:mn>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>200</mml:mn>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>200</mml:mn>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where <inline-formula id="inf76">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the input value of the <inline-formula id="inf77">
<mml:math id="m84">
<mml:mrow>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in the input layer, <inline-formula id="inf78">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight between <inline-formula id="inf79">
<mml:math id="m86">
<mml:mrow>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in the input layer and <inline-formula id="inf80">
<mml:math id="m87">
<mml:mrow>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in the hidden layer 1, <inline-formula id="inf81">
<mml:math id="m88">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight between <inline-formula id="inf82">
<mml:math id="m89">
<mml:mrow>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in hidden layer 1 and <inline-formula id="inf83">
<mml:math id="m90">
<mml:mrow>
<mml:msup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in hidden layer 2, and <inline-formula id="inf84">
<mml:math id="m91">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the weight between <inline-formula id="inf85">
<mml:math id="m92">
<mml:mrow>
<mml:msup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in hidden layer 2 and <inline-formula id="inf86">
<mml:math id="m93">
<mml:mrow>
<mml:msup>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in the output layer (i.e., <inline-formula id="inf87">
<mml:math id="m94">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). <inline-formula id="inf88">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf89">
<mml:math id="m96">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf90">
<mml:math id="m97">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the biases of <inline-formula id="inf91">
<mml:math id="m98">
<mml:mrow>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in hidden layer 1, <inline-formula id="inf92">
<mml:math id="m99">
<mml:mrow>
<mml:msup>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in hidden layer 2, and <inline-formula id="inf93">
<mml:math id="m100">
<mml:mrow>
<mml:msup>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> node in the output layer, respectively.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Architecture of the ANN-based model consists of an input layer, two hidden layers with 200 nodes (N), and an output layer. The weights between nodes (<italic>w</italic>) are illustrated.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g006.tif"/>
</fig>
<p>Each node in an input layer receives an independent variable (e.g., the N-value). At each node, the value (<inline-formula id="inf94">
<mml:math id="m101">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is multiplied by the corresponding weight (<inline-formula id="inf95">
<mml:math id="m102">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), which is summed with other values multiplied by other weights in the same layer. A bias (<inline-formula id="inf96">
<mml:math id="m103">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is then added to the sum of all values multiplied by each weight in the layer. The bias provides a better generalization ability to the model by enhancing the fitting flexibility. Then, the activation function (<inline-formula id="inf97">
<mml:math id="m104">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) is applied to the sum of <inline-formula id="inf98">
<mml:math id="m105">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf99">
<mml:math id="m106">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> gives a non-linear property to help the ANN model capture the complex data patterns. The activation function outputs a value that becomes the input value of the node for the next layer.</p>
<p>We applied a rectified linear unit (ReLU) to <inline-formula id="inf100">
<mml:math id="m107">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf101">
<mml:math id="m108">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, where the ReLU is a widely used non-linear activation function (<xref ref-type="bibr" rid="B10">Boob et al., 2020</xref>). Specifically, the ReLU is defined as <inline-formula id="inf102">
<mml:math id="m109">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf103">
<mml:math id="m110">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if <inline-formula id="inf104">
<mml:math id="m111">
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula id="inf105">
<mml:math id="m112">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">X</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> if <inline-formula id="inf106">
<mml:math id="m113">
<mml:mrow>
<mml:mi mathvariant="normal">X</mml:mi>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Model training strategy</title>
<p>Before training the model, the categorical variables (i.e., geology and soil/rock type) needed to be transformed into numerical variables. We mapped the variables into integers, which were then encoded in a binary format. This method is called binary encoding, which has been popularly utilized in applications (e.g., <xref ref-type="bibr" rid="B27">Jackson and Agrawal, 2019</xref>; <xref ref-type="bibr" rid="B60">Yousef et al., 2019</xref>). Here is an example using the soil/rock type in the K-NET dataset, which includes 12 unique features (i.e., 12 IDs). First, the length of the encoding vector was determined as <inline-formula id="inf107">
<mml:math id="m114">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>log</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>12</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Second, each ID is converted into binary format, e.g., &#x2018;sandy soil&#x2019; (ID &#x3d; 1) to [0, 0, 0, 1], &#x2018;fill soil&#x2019; (ID &#x3d; 6) to [0, 1, 1, 0], and &#x2018;volcanic ash clay&#x2019; (ID &#x3d; 12) to [1, 1, 0, 0]. Each bit number in the vector, for example, 1, 1, 0, and 0 in &#x2018;volcanic ash clay&#x2019; (ID &#x3d; 12) work as independent variables. Using this method, the total number of input variables was increased from 9 to 18 for the K-NET dataset and from 7 to 22 for the KiK-net dataset.</p>
<p>We applied the five-fold cross-validation (CV), which has been widely utilized in model evaluation (<xref ref-type="bibr" rid="B8">Berrar, 2019</xref>). This approach assesses the generalization ability of models and prevents overfitting. The five-fold CV divides the entire dataset randomly into five roughly equal folds. Then, the model uses four folds for training and the remaining one fold for testing (i.e., 80% for training and 20% for test dataset). We repeated for five times: i.e., we developed five ML models. The test results from these five experiments were aggregated to evaluate the general performance of the ML algorithm.</p>
<p>This study aims to train ML models using the data for some sites and evaluate the model performance using the data for new sites. Therefore, all data samples were split based on site locations and not on whole data samples. Each fold is allocated 20% of the total sites but may not be divided exactly. With our case as an example, the K-NET sites were divided into training and testing parts as follows: 797:199 (for four experiments), and 796:200 (for one experiment). For the KiK-net dataset, the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> data were divided into 541:136 (for two experiments) and 542:135 (for three experiments), and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> data were separated equally for all experiments: 540 for training and 135 for testing.</p>
</sec>
<sec id="s5">
<title>5 Validation</title>
<sec id="s5-1">
<title>5.1 Comparison between predictions and measurements</title>
<p>The three ML-based models developed in this study were evaluated for each test fold after training. <xref ref-type="fig" rid="F7">Figure 7</xref> presents the measured wave velocities (i.e., <inline-formula id="inf108">
<mml:math id="m115">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf109">
<mml:math id="m116">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) versus the estimated values (<inline-formula id="inf110">
<mml:math id="m117">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>; <inline-formula id="inf111">
<mml:math id="m118">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) using the three models for the entire K-NET test data samples. Note that these data were aggregated from five test folds from the five experiments (i.e., five derived ML models). As a performance indicator for <italic>V</italic>
<sub>
<italic>S</italic>
</sub> prediction, we calculated the root mean squared error (<inline-formula id="inf112">
<mml:math id="m119">
<mml:mrow>
<mml:msub>
<mml:mtext>RMSE</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) as:<disp-formula id="e8">
<mml:math id="m120">
<mml:mrow>
<mml:msub>
<mml:mtext>RMSE</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi mathvariant="normal">s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf113">
<mml:math id="m121">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the estimated <inline-formula id="inf114">
<mml:math id="m122">
<mml:mrow>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> value for the <inline-formula id="inf115">
<mml:math id="m123">
<mml:mrow>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> data sample, <inline-formula id="inf116">
<mml:math id="m124">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the corresponding measurement, and <inline-formula id="inf117">
<mml:math id="m125">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the data sample size. The RMSE values calculated for each of the five test folds were averaged. The <inline-formula id="inf118">
<mml:math id="m126">
<mml:mrow>
<mml:msub>
<mml:mtext>RMSE</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was computed in the same manner. The <inline-formula id="inf119">
<mml:math id="m127">
<mml:mrow>
<mml:msub>
<mml:mtext>RMSE</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf120">
<mml:math id="m128">
<mml:mrow>
<mml:msub>
<mml:mtext>RMSE</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the smallest for the GB-based model and the largest for the ANN-based model.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Measured velocity values (i.e., <inline-formula id="inf121">
<mml:math id="m129">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf122">
<mml:math id="m130">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) versus velocity values estimated by the three ML-based models (i.e., <inline-formula id="inf123">
<mml:math id="m131">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf124">
<mml:math id="m132">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) for the K-NET. The 1:1 lines are depicted as red dashed lines. The color bar on the right side represents the data density. The data were aggregated from five test folds from the five experiments (i.e., five derived ML models).</p>
</caption>
<graphic xlink:href="feart-11-1267386-g007.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F8">Figure 8</xref> presents the measured wave velocities (i.e., <inline-formula id="inf125">
<mml:math id="m133">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf126">
<mml:math id="m134">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) versus the estimated values (<inline-formula id="inf127">
<mml:math id="m135">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>; <inline-formula id="inf128">
<mml:math id="m136">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) using the three models for the KiK-net dataset. <inline-formula id="inf129">
<mml:math id="m137">
<mml:mrow>
<mml:msub>
<mml:mtext>RMSE</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf130">
<mml:math id="m138">
<mml:mrow>
<mml:msub>
<mml:mtext>RMSE</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the smallest for the GB-based model indicating a stronger alignment along the 1:1 line, and the largest for the RF-based model. The results for individual experiments are included in <xref ref-type="sec" rid="s12">Supplementary Appendix &#x2160;</xref> of the Electronic Supplement.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Measured velocity values (i.e., <inline-formula id="inf131">
<mml:math id="m139">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf132">
<mml:math id="m140">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) versus velocity values estimated by the three ML-based models (i.e., <inline-formula id="inf133">
<mml:math id="m141">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf134">
<mml:math id="m142">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) for the KiK-net. The 1:1 lines are depicted as red dashed lines. The color bar on the right side represents the data density. The data were aggregated from five test folds from the five experiments (i.e., five derived ML models).</p>
</caption>
<graphic xlink:href="feart-11-1267386-g008.tif"/>
</fig>
<p>The RMSE depends on the study area and data features including the number of sites and velocities distribution. Many previous studies have utilized varying ranges of <italic>V</italic>
<sub>
<italic>S</italic>
</sub> to make predictions for different geological regions, resulting in varied RMSEs. For example, <xref ref-type="bibr" rid="B6">Ataee et al. (2019)</xref> utilized uncorrected and corrected SPT-N with 88 boreholes to predict <italic>V</italic>
<sub>
<italic>S</italic>
</sub>. The results using uncorrected SPT-N and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> under approximately 1,200&#xa0;m/s presented that the RMSEs of the models ranged from 94.512 to 104.149&#xa0;m/s. Those using corrected SPT-N and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> under approximately 600&#xa0;m/s presented RMSEs ranging from 59.423 to 67.473&#xa0;m/s. <xref ref-type="bibr" rid="B23">Ghorbani et al. (2012)</xref> utilized corrected SPT blow counts, and effective overburden stress to predict <italic>V</italic>
<sub>
<italic>S</italic>
</sub>. They used 80 boreholes, where the <italic>V</italic>
<sub>
<italic>S</italic>
</sub> ranges from 66 to 363&#xa0;m/s. The RMSE of the prediction model is 37.2&#xa0;m/s. <xref ref-type="bibr" rid="B55">Sun et al. (2013)</xref> used tip resistance, sleeve friction, pore pressure, and overburden effective stress to establish the correlation with <italic>V</italic>
<sub>
<italic>S</italic>
</sub>. They utilized 17 sites, where the measured <italic>V</italic>
<sub>
<italic>S</italic>
</sub> is under approximately 400&#xa0;m/s. The RMSEs of the correlation forms are from 30.42 to 38.57&#xa0;m/s. Furthermore, <xref ref-type="bibr" rid="B16">Dumke and Berndt (2019)</xref> used 38 types of variables (e.g., depth below seafloor, surface heat flow, and distance to nearest spreading ridge) to predict <italic>V</italic>
<sub>
<italic>P</italic>
</sub>. They used 333 sites containing <italic>V</italic>
<sub>
<italic>P</italic>
</sub> above 4,000&#xa0;m/s, where the velocity range is not mentioned. The RMSEs vary approximately between 400 and 500&#xa0;m/s depending on the considered variables. The RMSE presented in this paper may be reasonable, given that the prediction models were made and tested for the larger number of sites distributed throughout Japan, which includes various study areas and a wider range of velocities than other studies. However, discrepancies have been observed, especially for the KiK-net: <italic>V</italic>
<sub>
<italic>P</italic>
</sub> dataset, implying that more region-specific depth-related variables may be needed to infer the velocity profiles better.</p>
<p>We further investigated the relationship between the measured and estimated velocities by employing the Regression Error Characteristic (REC) curve (<xref ref-type="bibr" rid="B9">Bi and Bennett, 2003</xref>). The REC curve depicts the relationship between the specified deviation tolerance on the <italic>x</italic>-axis, which is the error tolerance, and the <italic>y</italic>-axis for the proportion of data with prediction deviations smaller than the corresponding deviation. The resulting curve provides an estimation of the cumulative distribution function of the error. Furthermore, the REC curve quantifies the performance of the model by computing the area under the curve (AUC). A higher AUC value indicates better model performance. <xref ref-type="fig" rid="F9">Figure 9</xref> illustrates the REC curves for each model. The curves were individually computed for the five test folds from the five experiments and were then averaged to make a single curve. The AUC was subsequently calculated based on the single curve, representing the general performance of each model on the dataset. The results for K-NET and KiK-net, both at <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub>, reveal that all AUCs are above 0.702 for the specified deviations ranging from 0 to 1.0. Notably, the GB-based model has the highest AUC across all cases, indicating its relatively strong predictive performance within the deviation range.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>REC curves for individual models: <bold>(A)</bold> <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <bold>(B)</bold> <italic>V</italic>
<sub>
<italic>S</italic>
</sub> for K-NET, and <bold>(C)</bold> <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <bold>(D)</bold> <italic>V</italic>
<sub>
<italic>S</italic>
</sub> for KiK-net. Each curve represents the average accuracy across the five test folds from the five experiments (i.e., five derived ML models), with the specified deviation.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g009.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F10">Figure 10</xref> presents maps of RMSEs for the GB-based model for all the sites considered in this study, which were aggregated from five test folds from the five experiments (i.e., five derived ML models). Overall, the models for <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> of the K-NET sites (<xref ref-type="fig" rid="F10">Figures 10A, B</xref>) indicate that almost 80% of the sites have RMSE values within the range of (0, 500] for <italic>V</italic>
<sub>
<italic>P</italic>
</sub>, and almost 99% are within the same range for <italic>V</italic>
<sub>
<italic>S</italic>
</sub>. In contrast, for KiK-net (<xref ref-type="fig" rid="F10">Figures 10C, D</xref>), approximately 46% of the sites have RMSE values within the (500, 1,000] range for <italic>V</italic>
<sub>
<italic>P</italic>
</sub>, and about 63% are within the (0, 500] range for <italic>V</italic>
<sub>
<italic>S</italic>
</sub>. It can be noticed that the RMSE values larger than 1,000&#xa0;m/s for the estimated <italic>V</italic>
<sub>
<italic>P</italic>
</sub> values at the K-NET sites are concentrated in the area around 139 &#xb0;E and 35.5 &#xb0;N (<xref ref-type="fig" rid="F10">Figure 10A</xref>). Furthermore, the RMSE values greater than 1,500&#xa0;m/s and 1,000&#xa0;m/s for the estimated <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> values, respectively, at the KiK-net sites are mainly clustered in the region around 137 &#xb0;E and 35 &#xb0;N (<xref ref-type="fig" rid="F10">Figures 10C, D</xref>). The RMSEs for the KiK-net sites show a certain pattern along the east coast (from 140 to 142 &#xb0;E and from 36 to 40 &#xb0;N) (see <xref ref-type="fig" rid="F10">Figure 10D</xref>). These observations imply that there could be factors that can affect the <italic>V</italic>
<sub>
<italic>S</italic>
</sub> and <italic>V</italic>
<sub>
<italic>P</italic>
</sub> values, other than those considered in this study.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Maps for RMSE values of the GB-based model for <bold>(A)</bold> <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <bold>(B)</bold> <italic>V</italic>
<sub>
<italic>S</italic>
</sub> of all of the K-NET sites, and those for <bold>(C)</bold> <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <bold>(D)</bold> <italic>V</italic>
<sub>
<italic>S</italic>
</sub> of all of the KiK-net sites. The data for the sites were aggregated from five test folds of the five experiments (i.e., five derived ML models). The count numbers of the color-coded RMSE ranges are presented inside each of the panels.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g010.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F11">Figure 11</xref> shows examples of the wave velocity profiles predicted by the GB-based model compared with the measured profiles at the nine K-NET sites. The eight and one sample profiles were randomly selected from the <italic>V</italic>
<sub>
<italic>S</italic>
</sub> RMSE bands of (0, 500] m/s and (500, 1,000] m/s, respectively, from the entire test folds from five experiments. The wave velocities predicted for the HKD024, FKO015, and KNG008 sites (<xref ref-type="fig" rid="F11">Figures 11A&#x2013;C</xref>, respectively) are in good agreement with the measured profiles when compared to the other illustration, producing RMSE values &#x2264;186&#xa0;m/s. In contrast, there are some discrepancies between the measured and predicted profiles at certain depth ranges for the rest of the sites. At GNM014, <italic>V</italic>
<sub>
<italic>P</italic>
</sub> is overestimated at depths of up to 9&#xa0;m (<xref ref-type="fig" rid="F11">Figure 11D</xref>). At NGN025, the wave velocities are underestimated at depths greater than 13&#xa0;m (<xref ref-type="fig" rid="F11">Figure 11E</xref>). At EHM012, <italic>V</italic>
<sub>
<italic>P</italic>
</sub> is underestimated at depths from 2 to 6&#xa0;m (<xref ref-type="fig" rid="F11">Figure 11F</xref>). There are also some discrepancies in the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> profiles at the YMT006, IBR009, and KGW008 sites (<xref ref-type="fig" rid="F11">Figures 11G&#x2013;I</xref>, respectively). In detail, the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> is consistently overestimated across the entire depth range at the YMT006 site (<xref ref-type="fig" rid="F11">Figure 11G</xref>). At the IBR009 site, <italic>V</italic>
<sub>
<italic>P</italic>
</sub> is underestimated up to a depth of 5&#xa0;m (<xref ref-type="fig" rid="F11">Figure 11H</xref>). Similarly, the KGW008 site demonstrates underestimation up to 3&#xa0;m and overestimation at depths beyond 10&#xa0;m (<xref ref-type="fig" rid="F11">Figure 11I</xref>).</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Measured wave velocity profiles (<inline-formula id="inf135">
<mml:math id="m143">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf136">
<mml:math id="m144">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) versus the velocity profiles predicted by the GB-based model (<inline-formula id="inf137">
<mml:math id="m145">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf138">
<mml:math id="m146">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) at the nine K-NET sites <bold>(A&#x2013;I)</bold> from five test folds.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g011.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F12">Figure 12</xref> presents examples of the wave velocity profiles estimated by the GB-based model compared with the measured profiles at the nine KiK-net sites. The six, two, and one sample profiles were randomly selected from the <italic>V</italic>
<sub>
<italic>S</italic>
</sub> RMSE bands of (0, 500], (500, 1,000], and (1,500, 2000], respectively, from the entire test folds from five experiments. The estimated wave velocities for the HRSH15, KOCH11, IBRH12, and TCGH15 sites (<xref ref-type="fig" rid="F12">Figures 12A&#x2013;C, E</xref>, respectively) comparatively match well with the measured profiles, producing RMSE values &#x2264;388&#xa0;m/s. Some discrepancies are observed at specific depth ranges for the other sites. At IBRH07, the wave velocities are overestimated at depths from 51&#xa0;m to 650&#xa0;m and underestimated at depths from 651&#xa0;m to 1,050&#xa0;m (<xref ref-type="fig" rid="F12">Figure 12D</xref>); however, the estimated wave velocities show relatively close agreement beyond 1,050&#xa0;m. At SITH03, the velocities are overestimated almost throughout the depth (<xref ref-type="fig" rid="F12">Figure 12F</xref>). At AICH12, the velocities are underestimated at depths greater than 49&#xa0;m (<xref ref-type="fig" rid="F12">Figure 12G</xref>). Overestimations are observed at the ISKH06 site for depths greater than 44&#xa0;m (<xref ref-type="fig" rid="F12">Figure 12H</xref>). For the YMTH08 site, velocities are overestimated for depths exceeding 18&#xa0;m, while the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> is underestimated for depths up to 6&#xa0;m (<xref ref-type="fig" rid="F12">Figure 12I</xref>).</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>Measured wave velocity profiles (<inline-formula id="inf139">
<mml:math id="m147">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf140">
<mml:math id="m148">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) versus the velocity profiles predicted by the GB-based model (<inline-formula id="inf141">
<mml:math id="m149">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf142">
<mml:math id="m150">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>) at the nine KiK-net sites <bold>(A&#x2013;I)</bold> from five test folds.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g012.tif"/>
</fig>
<p>Some discrepancies were observed in relation to a certain depth and profile patterns, as shown in <xref ref-type="fig" rid="F11">Figure 11</xref> and <xref ref-type="fig" rid="F12">Figure 12</xref>. It is possible that the model could not well predict velocities for sites that have unusual profile patterns or for those that have not been frequently used when training or are not included in the training dataset. The model was trained to reduce the overall error for the entire sites used in training, so it might not well generalize the unseen patterns. As seen in the samples in <xref ref-type="fig" rid="F11">Figure 11</xref> and <xref ref-type="fig" rid="F12">Figure 12</xref>, the model was trained to predict slower velocities near the ground surface and faster velocities at greater depths. Furthermore, model predicts velocities gradually increasing, and does not predict well the abrupt velocity changes (e.g., <xref ref-type="fig" rid="F11">Figure 11E</xref>; <xref ref-type="fig" rid="F12">Figure 12G</xref>). There are velocity reversals at depths greater than 44&#xa0;m at ISKH06 (<xref ref-type="fig" rid="F12">Figure 12H</xref>), and the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> values are unusually faster near the ground surface at YMTH08 (<xref ref-type="fig" rid="F12">Figure 12I</xref>). It turned out that the model was not able to capture these profiles.</p>
<p>For the systematic evaluation of discrepancies between measured and estimated wave velocities, we computed the residuals for all the continuous variables as<disp-formula id="e9">
<mml:math id="m151">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <inline-formula id="inf143">
<mml:math id="m152">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf144">
<mml:math id="m153">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are the residuals for <italic>V</italic>
<sub>
<italic>S</italic>
</sub> and <italic>V</italic>
<sub>
<italic>P</italic>
</sub>, respectively. We also calculated the standard deviations of the residuals and biases (i.e., mean values of the residuals) for all trained models.</p>
<p>
<xref ref-type="fig" rid="F13">Figure 13</xref> shows the <inline-formula id="inf145">
<mml:math id="m154">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> of the GB-based model for the K-NET stations, which were aggregated from five test folds from the five experiments. The <inline-formula id="inf146">
<mml:math id="m155">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> ranges from &#x2212;1.798 to 1.677 and is aligned along zeros with respect to all continuous variables with a small bias (i.e., &#x2212;0.066). <xref ref-type="fig" rid="F14">Figure 14</xref> shows the <inline-formula id="inf147">
<mml:math id="m156">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> of the GB-based model for the K-NET stations, which were aggregated from five test folds from the five experiments. The <inline-formula id="inf148">
<mml:math id="m157">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> ranges from &#x2212;1.765 to 1.396 and is also aligned along with zeros with a bias value of &#x2212;0.059. The standard deviation values for <inline-formula id="inf149">
<mml:math id="m158">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are 0.354, 0.358, and 0.374 for the GB-, RF-, and ANN-based models, respectively. The standard deviation values for <inline-formula id="inf150">
<mml:math id="m159">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are 0.360, 0.361, and 0.383 for the GB-, RF-, and ANN-based models, respectively. The residuals do not show any trend with the considered variables, indicating that all the variables have certain contributions to the model, or that some of the variables do not have influence on wave velocities. Moreover, the results seem reasonable when compared to those of <xref ref-type="bibr" rid="B36">Kwak et al. (2015)</xref>, who presented the range of standard deviation of residuals for <italic>V</italic>
<sub>
<italic>S</italic>
</sub> prediction models made using K-NET for each soil/rock type, which was from 0.245 to 0.462. Because the <inline-formula id="inf151">
<mml:math id="m160">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf152">
<mml:math id="m161">
<mml:mrow>
<mml:msup>
<mml:mtext>Res</mml:mtext>
<mml:msub>
<mml:mi>V</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> of the GB-based model for KiK-net stations show the same aspects, we describe the results in <xref ref-type="sec" rid="s12">Supplementary Appendix &#x2161;</xref> of the Electronic Supplement.</p>
<fig id="F13" position="float">
<label>FIGURE 13</label>
<caption>
<p>Residuals of the <italic>V</italic>
<sub>
<italic>S</italic>
</sub> estimated by the GB-based model for the K-NET stations with respect to all the continuous variables: <bold>(A)</bold> depth; <bold>(B)</bold> elevation; <bold>(C)</bold> slope angle; <bold>(D)</bold> N-value; and <bold>(E)</bold> density. The data were aggregated from five test folds from the five experiments (i.e., five derived ML models).</p>
</caption>
<graphic xlink:href="feart-11-1267386-g013.tif"/>
</fig>
<fig id="F14" position="float">
<label>FIGURE 14</label>
<caption>
<p>Residuals of the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> estimated by the GB-based model for the K-NET stations with respect to all the continuous variables: <bold>(A)</bold> depth; <bold>(B)</bold> elevation; <bold>(C)</bold> slope angle; <bold>(D)</bold> N-value; and <bold>(E)</bold> density. The data were aggregated from five test folds from the five experiments (i.e., five derived ML models).</p>
</caption>
<graphic xlink:href="feart-11-1267386-g014.tif"/>
</fig>
</sec>
<sec id="s5-2">
<title>5.2 Variable importance</title>
<p>We examined the contribution levels of the independent variables to the prediction accuracy of the best model, the GB-based model. The method is called variable importance (VI), which is computed as the sum of the decrease in error when a variable splits a tree node (e.g., a node split by an N-value <inline-formula id="inf153">
<mml:math id="m162">
<mml:mrow>
<mml:mo>&#x2264;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 14.75). The variable importance for variable <italic>x</italic> (i.e., VI(<italic>x</italic>)) is calculated as follows:<disp-formula id="e10">
<mml:math id="m163">
<mml:mrow>
<mml:mtext>VI</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>B</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2206;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>where <inline-formula id="inf154">
<mml:math id="m164">
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the tree index (<inline-formula id="inf155">
<mml:math id="m165">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf156">
<mml:math id="m166">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>), <inline-formula id="inf157">
<mml:math id="m167">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the node in a specific tree model (<italic>T</italic>
<sub>
<italic>b</italic>
</sub>), and <inline-formula id="inf158">
<mml:math id="m168">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the variable used for splitting the node in which <inline-formula id="inf159">
<mml:math id="m169">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the splitting criterion (<inline-formula id="inf160">
<mml:math id="m170">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) at note <italic>t</italic>. <inline-formula id="inf161">
<mml:math id="m171">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the proportion (<inline-formula id="inf162">
<mml:math id="m172">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>) of data samples reaching <inline-formula id="inf163">
<mml:math id="m173">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf164">
<mml:math id="m174">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of total training data samples, and <inline-formula id="inf165">
<mml:math id="m175">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of data samples at node <inline-formula id="inf166">
<mml:math id="m176">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf167">
<mml:math id="m177">
<mml:mrow>
<mml:mo>&#x2206;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the impurity reduction at node <inline-formula id="inf168">
<mml:math id="m178">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which can be expressed as follows:<disp-formula id="e11">
<mml:math id="m179">
<mml:mrow>
<mml:mo>&#x2206;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf169">
<mml:math id="m180">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the MSE at node <inline-formula id="inf170">
<mml:math id="m181">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf171">
<mml:math id="m182">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf172">
<mml:math id="m183">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are the MSEs at the left child node <inline-formula id="inf173">
<mml:math id="m184">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> and right child node <inline-formula id="inf174">
<mml:math id="m185">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, respectively, split from node <inline-formula id="inf175">
<mml:math id="m186">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf176">
<mml:math id="m187">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf177">
<mml:math id="m188">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the numbers of data samples at <inline-formula id="inf178">
<mml:math id="m189">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>l</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf179">
<mml:math id="m190">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, respectively.</p>
<p>
<xref ref-type="fig" rid="F15">Figures 15A, B</xref> show the relative VIs for the K-NET independent variables for the GB-based models. The relative VI was computed by VI for each variable divided by the total VI for all variables. The VI was calculated on each test fold, and the VIs for all the five test folds were averaged. The VIs computed for binary codes were summed for the categorical variables. Three depth-dependent variables (i.e., depth, N-value, and density) have the highest VIs for both <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> models. The depth is ranked at the top for the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> model, whereas the N-value is ranked at the top for the <italic>V</italic>
<sub>
<italic>S</italic>
</sub> model. <xref ref-type="fig" rid="F15">Figures 15C, D</xref> present the relative VIs for the KiK-net dataset. The depth turned out to be the most critical variable for both <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> models. The effect of the site location is more significant for the KiK-net model than for the K-NET model. The slope angle and elevation have a certain influence on the models, whereas the soil/rock type and geology have the least influence. Although the influence of the geology turned out to be insignificant, the performance of the GB-based model was enhanced by including it. The RMSEs of the model were reduced from 615&#xa0;m/s to 597&#xa0;m/s for <italic>V</italic>
<sub>
<italic>S</italic>
</sub>, and from 979&#xa0;m/s to 961&#xa0;m/s for <italic>V</italic>
<sub>
<italic>P</italic>
</sub>, implying that it is also related to wave velocities at a deeper depth.</p>
<fig id="F15" position="float">
<label>FIGURE 15</label>
<caption>
<p>Relative variable importance (VI) for the GB-based models for <bold>(A)</bold> K-NET (<italic>V</italic>
<sub>
<italic>P</italic>
</sub>), <bold>(B)</bold> K-NET (<italic>V</italic>
<sub>
<italic>S</italic>
</sub>), <bold>(C)</bold> KiK-net (<italic>V</italic>
<sub>
<italic>P</italic>
</sub>), and <bold>(D)</bold> KiK-net (<italic>V</italic>
<sub>
<italic>S</italic>
</sub>). The variables are presented in an order of descending relative VI.</p>
</caption>
<graphic xlink:href="feart-11-1267386-g015.tif"/>
</fig>
<p>The confining pressure increases with depth, leading to an increase in the density, N-value, and wave velocities. Therefore, the depth and associated variables were determined to be most strongly correlated, as revealed by VI. The slope and elevation are related with shear stiffnesses, which eventually affect wave velocities. The site coordinates are relatively high VI, implying that they may be associated with site conditions that were not captured by other variables. The geology has the lowest VI, as it is for the ground surface. However, we included it in the model because of its certain effect in enhancing the predictive performance.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>This paper presented three ML-based models (i.e., GB-, RF-, and ANN-based models) predicting <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> in Japan. We used borehole databases from the two seismograph networks, K-NET and KiK-net. We considered various factors such as depth, N-value, density, slope angle, elevation, geology, soil/rock type, and site coordinates. The number of trees was designated as 100 to train the RF- and GB-based models. We developed an ANN-based model with four layers, where each hidden layer included 200 nodes.</p>
<p>The models were trained and evaluated on the datasets using the five-fold cross-validation. The average RMSEs across all test folds showed that the GB-based model provided the best estimation among the other models for both K-NET and KiK-net sites. The RMSEs of the GB-based model for <italic>V</italic>
<sub>
<italic>S</italic>
</sub> and <italic>V</italic>
<sub>
<italic>P</italic>
</sub> of the K-NET sites were 146 and 437&#x00A0;m/s, respectively, and those of the KiK-net sites were 597 and 961&#x00A0;m/s, respectively, while those of the other models ranging from 150 to 462&#x00A0;m/s for K-NET and from 659 to 1,116&#x00A0;m/s for KiK-net. Furthermore, the REC curve indicated that the GB-based model revealed relatively high performance within the deviation range. We also validated the GB-based model by checking the residuals between the measured and estimated wave velocities with respect to various variables. The variable importance of the model for K-NET indicated that depth, N-value, and density were the essential variables in predicting the <italic>V</italic>
<sub>
<italic>P</italic>
</sub> and <italic>V</italic>
<sub>
<italic>S</italic>
</sub> of the K-NET sites. Note that we used the unnormalized N-values for the K-NET sites, which might lower the prediction capability. For KiK-net sites, depth was the most influential variable. The site longitude also had a high relative variable importance value, indicating the roles of factors other than those considered in this study. The geology has the smallest VI values, as shown in <xref ref-type="fig" rid="F15">Figure 15</xref>. However, it turned out that including the geology can improve the model performance, decreasing the RMSE values. In addition, we consider that including latitude and longitude is necessary because these improved prediction performances of the models, capturing the effects that were not captured by other variables.</p>
<p>This paper proposed a model for predicting wave velocities based on various factors, which can be used for site exploration in various fields, including rock engineering and petroleum engineering. The key findings of this study highlight that common machine learning algorithms can reasonably predict the wave velocity profiles across the region of Japan as an example. The results from cross-validation present the general performances of models on the dataset and site-specific performances specifically for the GB-based model. The study reveals the importance of input variables contributing to predicting accuracy. Moreover, it suggests that considering more region-specific variables including site coordinates can assist the models in interpreting complicated relationships.</p>
<p>As for the limitations of this study, the models are limited by their reliance on borehole databases exclusively obtained from specific seismograph networks in Japan. This approach may present a bias towards the conditions within these networks. Consequently, predictive performance could be constrained when extending the applications to regions with different geological attributes. In this context, ensuring the consistency of the environmental and experimental conditions, and the employed measured data is crucial to guarantee the validity of results beyond the area considered in this study. Additionally, even though the various variables were included, an incomplete representation remains for specific regions. This implies the presence of intricate geological properties that necessitate analysis to understand their influence on the prediction of wave velocities in a particular area. Furthermore, the study reveals that incorporating site coordinates can influence predictive performance. Nevertheless, the specific contributions of these variables to predictive performance concerning geological characteristics remain subject to consideration. While this study confirmed that the most commonly used machine-learning techniques could be successfully applied for predicting wave velocities, exploring more advanced techniques and investigating additional factors in the future will enhance the prediction performance.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://www.kyoshin.bosai.go.jp/">https://www.kyoshin.bosai.go.jp/</ext-link>, Strong-motion Seismograph Networks (K-NET, KiK-net).</p>
</sec>
<sec id="s8">
<title>Author contributions</title>
<p>JK: Conceptualization, Data curation, Formal Analysis, Methodology, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. J-DK: Conceptualization, Data curation, Formal Analysis, Writing&#x2013;original draft. BK: Conceptualization, Formal Analysis, Funding acquisition, Project administration, Supervision, Writing&#x2013;original draft, Writing&#x2013;review and editing.</p>
</sec>
<sec id="s9">
<title>Funding</title>
<p>This work was supported by KOREA HYDRO &#x0026; NUCLEAR POWER CO., LTD (No. 2022-Tech-03) and the Korea Agency for Infrastructure Technology Advancement (KAIA) grant funded by the Ministry of Land, Infrastructure, and Transport (Grant 21CATAP-C164148-01). The funder was not involved in the study design, collection, analysis, interpretation of data, the writing of this article, or the decision to submit it for publication. The opinions, findings, and conclusions or recommendations expressed in this article are solely those of the authors and do not represent those of the funders.</p>
</sec>
<ack>
<p>We are indebted to the National Research Institute for Earth Science and Disaster Resilience (NIED), Japan, for making the resources of K-NET and KiK-net seismographs publicly available.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/feart.2023.1267386/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/feart.2023.1267386/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Abadi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Agarwal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Barham</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Brevdo</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Citro</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <source>Tensorflow: large-scale machine learning on heterogeneous distributed systems</source>. <comment>arXiv preprint arXiv:1603.04467</comment>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Akin</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Kramer</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Topal</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Empirical correlations of shear wave velocity (Vs) and penetration resistance (SPT-N) for different soils in an earthquake-prone area (Erbaa-Turkey)</article-title>. <source>Eng. Geol.</source> <volume>119</volume> (<issue>1-2</issue>), <fpage>1</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.1016/j.enggeo.2011.01.007</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ameen</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Smart</surname>
<given-names>B. G.</given-names>
</name>
<name>
<surname>Somerville</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Hammilton</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Naji</surname>
<given-names>N. A.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Predicting rock mechanical properties of carbonates from wireline logs (A case study: arab-D reservoir, ghawar field, Saudi arabia)</article-title>. <source>Mar. Petroleum Geol.</source> <volume>26</volume> (<issue>4</issue>), <fpage>430</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1016/j.marpetgeo.2009.01.017</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andrus</surname>
<given-names>R. D.</given-names>
</name>
<name>
<surname>Piratheepan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ellis</surname>
<given-names>B. S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Juang</surname>
<given-names>C. H.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Comparing liquefaction evaluation methods using penetration-V<sub>S</sub> relationships</article-title>. <source>Soil Dyn. Earthq. Eng.</source> <volume>24</volume> (<issue>9-10</issue>), <fpage>713</fpage>&#x2013;<lpage>721</lpage>. <pub-id pub-id-type="doi">10.1016/j.soildyn.2004.06.001</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anemangely</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ramezanzadeh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Amiri</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hoseinpour</surname>
<given-names>S.-A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Machine learning technique for the prediction of shear wave velocity using petrophysical logs</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>174</volume>, <fpage>306</fpage>&#x2013;<lpage>327</lpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2018.11.032</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ataee</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Moghaddas</surname>
<given-names>N. H.</given-names>
</name>
<name>
<surname>Lashkaripour</surname>
<given-names>G. R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Estimating shear wave velocity of soil using standard penetration test (SPT) blow counts in Mashhad city</article-title>. <source>J. Earth Syst. Sci.</source> <volume>128</volume> (<issue>3</issue>), <fpage>1</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1007/s12040-019-1077-x</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bajaj</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Anbazhagan</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Seismic site classification and correlation between V<sub>S</sub> and SPT-N for deep soil sites in Indo-Gangetic Basin</article-title>. <source>J. Appl. Geophys.</source> <volume>163</volume>, <fpage>55</fpage>&#x2013;<lpage>72</lpage>. <pub-id pub-id-type="doi">10.1016/j.jappgeo.2019.02.011</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berrar</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Cross-validation</article-title>. <source>Encycl. Bioinforma. Comput. Biol.</source> <volume>1</volume>, <fpage>542</fpage>&#x2013;<lpage>545</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-12-809633-8.20349-X</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bennett</surname>
<given-names>K. P.</given-names>
</name>
</person-group> (<year>2003</year>). &#x201c;<article-title>Regression error characteristic curves</article-title>,&#x201d; in <conf-name>Proceedings of the 20th international conference on machine learning (ICML-03))</conf-name>, <conf-loc>Washington, DC USA</conf-loc>, <conf-date>August 21 - 24, 2003</conf-date>, <fpage>43</fpage>&#x2013;<lpage>50</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boob</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Dey</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Complexity of training relu neural network</article-title>. <source>Discrete Optim.</source> <volume>2020</volume>, <fpage>100620</fpage>. <pub-id pub-id-type="doi">10.1016/j.disopt.2020.100620</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume> (<issue>1</issue>), <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zoback</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Khaksar</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Empirical relations between rock strength and physical properties in sedimentary rocks</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>51</volume> (<issue>3-4</issue>), <fpage>223</fpage>&#x2013;<lpage>237</lpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2006.01.003</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Czajkowski</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kretowski</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Decision tree underfitting in mining of gene expression data. An evolutionary multi-test tree approach</article-title>. <source>Expert Syst. Appl.</source> <volume>137</volume>, <fpage>392</fpage>&#x2013;<lpage>404</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2019.07.019</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Konat&#xe9;</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Support vector machine as an alternative method for lithology classification of crystalline rocks</article-title>. <source>J. Geophys. Eng.</source> <volume>14</volume> (<issue>2</issue>), <fpage>341</fpage>&#x2013;<lpage>349</lpage>. <pub-id pub-id-type="doi">10.1088/1742-2140/aa5b5b</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Di</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Investigation of the effects of fracture orientation and saturation on the Vp/Vs ratio and their implications</article-title>. <source>Rock Mech. Rock Eng.</source> <volume>52</volume> (<issue>9</issue>), <fpage>3293</fpage>&#x2013;<lpage>3304</lpage>. <pub-id pub-id-type="doi">10.1007/s00603-019-01770-3</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dumke</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Berndt</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Prediction of seismic P-wave velocity using machine learning</article-title>. <source>Solid earth.</source> <volume>10</volume> (<issue>6</issue>), <fpage>1989</fpage>&#x2013;<lpage>2000</lpage>. <pub-id pub-id-type="doi">10.5194/se-10-1989-2019</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Eberli</surname>
<given-names>G. P.</given-names>
</name>
<name>
<surname>Baechle</surname>
<given-names>G. T.</given-names>
</name>
<name>
<surname>Anselmetti</surname>
<given-names>F. S.</given-names>
</name>
<name>
<surname>Incze</surname>
<given-names>M. L.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Factors controlling elastic properties in carbonate sediments and rocks</article-title>. <source>Lead. Edge</source> <volume>22</volume> (<issue>7</issue>), <fpage>654</fpage>&#x2013;<lpage>660</lpage>. <pub-id pub-id-type="doi">10.1190/1.1599691</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fiorentino</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Quaranta</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Mylonakis</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lavorato</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Pagliaroli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Carlucci</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Seismic reassessment of the leaning tower of pisa: dynamic monitoring, site response, and SSI</article-title>. <source>Earthq. Spectra</source> <volume>35</volume> (<issue>2</issue>), <fpage>703</fpage>&#x2013;<lpage>736</lpage>. <pub-id pub-id-type="doi">10.1193/021518EQS037M</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Greedy function approximation: a gradient boosting machine</article-title>. <source>Ann. Statistics</source> <volume>29</volume>, <fpage>1189</fpage>&#x2013;<lpage>1232</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1013203451</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Stochastic gradient boosting</article-title>. <source>Comput. Statistics Data Analysis</source> <volume>38</volume> (<issue>4</issue>), <fpage>367</fpage>&#x2013;<lpage>378</lpage>. <pub-id pub-id-type="doi">10.1016/S0167-9473(01)00065-2</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<collab>Geological Survey of Japan</collab> (<year>2015</year>). <source>Seamless digital geological map of Japan 1: 200,000</source>. <publisher-loc>Japan</publisher-loc>: <publisher-name>National Institute of Advanced Industrial Science and Technology</publisher-name>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geurts</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Irrthum</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wehenkel</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Supervised learning with decision tree-based methods in computational and systems biology</article-title>. <source>Mol. Biosyst.</source> <volume>5</volume> (<issue>12</issue>), <fpage>1593</fpage>&#x2013;<lpage>1605</lpage>. <pub-id pub-id-type="doi">10.1039/B907946G</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghorbani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jafarian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Maghsoudi</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Estimating shear wave velocity of soil deposits using polynomial neural networks: application to liquefaction</article-title>. <source>Comput. Geosciences</source> <volume>44</volume>, <fpage>86</fpage>&#x2013;<lpage>94</lpage>. <pub-id pub-id-type="doi">10.1016/j.cageo.2012.03.002</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Harmon</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hashash</surname>
<given-names>Y. M.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Rathje</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Campbell</surname>
<given-names>K. W.</given-names>
</name>
<name>
<surname>Silva</surname>
<given-names>W. J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Site amplification functions for central and eastern north America&#x2013;Part II: modular simulation-based models</article-title>. <source>Earthq. Spectra</source> <volume>35</volume> (<issue>2</issue>), <fpage>815</fpage>&#x2013;<lpage>847</lpage>. <pub-id pub-id-type="doi">10.1193/091117EQS179M</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hasancebi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ulusay</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Empirical correlations between shear wave velocity and penetration resistance for ground shaking assessments</article-title>. <source>Bull. Eng. Geol. Environ.</source> <volume>66</volume> (<issue>2</issue>), <fpage>203</fpage>&#x2013;<lpage>213</lpage>. <pub-id pub-id-type="doi">10.1007/s10064-006-0063-0</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heath</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Wald</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Worden</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Thompson</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Smoczyk</surname>
<given-names>G. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A global hybrid V<sub>S30</sub> map with a topographic slope&#x2013;based default and regional map insets</article-title>. <source>Earthq. Spectra</source> <volume>36</volume> (<issue>3</issue>), <fpage>1570</fpage>&#x2013;<lpage>1584</lpage>. <pub-id pub-id-type="doi">10.1177/8755293020911137</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Jackson</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Performance Evaluation of different feature Encoding schemes on cybersecurity logs</article-title>,&#x201d; in <conf-name>2019 SoutheastCon.</conf-name>, <conf-date>11-14 April 2019</conf-date>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jamshidi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zamanian</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sahamieh</surname>
<given-names>R. Z.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>The effect of density and porosity on the correlation between uniaxial compressive strength and P-wave velocity</article-title>. <source>Rock Mech. Rock Eng.</source> <volume>51</volume> (<issue>4</issue>), <fpage>1279</fpage>&#x2013;<lpage>1286</lpage>. <pub-id pub-id-type="doi">10.1007/s00603-017-1379-8</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jena</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pradhan</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Almazroui</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Assiri</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>H.-J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Earthquake-induced liquefaction hazard mapping at national-scale in Australia using deep learning techniques</article-title>. <source>Geosci. Front.</source> <volume>14</volume> (<issue>1</issue>), <fpage>101460</fpage>.<pub-id pub-id-type="doi">10.1016/j.gsf.2022.101460</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jun</surname>
<given-names>M.-J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A comparison of a gradient boosting decision tree, random forests, and artificial neural networks to model urban land use changes: the case of the seoul metropolitan area</article-title>. <source>Int. J. Geogr. Inf. Sci.</source> <volume>35</volume> (<issue>11</issue>), <fpage>2149</fpage>&#x2013;<lpage>2167</lpage>. <pub-id pub-id-type="doi">10.1080/13658816.2021.1887490</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karthikeyan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Samui</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Application of statistical learning algorithms for prediction of liquefaction susceptibility of soil based on shear wave velocity</article-title>. <source>Geomatics, Nat. Hazards Risk</source> <volume>5</volume> (<issue>1</issue>), <fpage>7</fpage>&#x2013;<lpage>25</lpage>. <pub-id pub-id-type="doi">10.1080/19475705.2012.757252</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Mapping of ground motion amplifications for the fraser river delta in greater vancouver, Canada</article-title>. <source>Earthq. Eng. Eng. Vib.</source> <volume>18</volume> (<issue>4</issue>), <fpage>703</fpage>&#x2013;<lpage>717</lpage>. <pub-id pub-id-type="doi">10.1007/s11803-019-0531-8</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hwang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Seo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Ground motion amplification models for Japan using machine learning techniques</article-title>. <source>Soil Dyn. Earthq. Eng.</source> <volume>132</volume>, <fpage>106095</fpage>. <pub-id pub-id-type="doi">10.1016/j.soildyn.2020.106095</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kottke</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Hashash</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Moss</surname>
<given-names>C. J.</given-names>
</name>
<name>
<surname>Nikolaou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rathje</surname>
<given-names>E. M.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). &#x201c;<article-title>Development of geologic site classes for seismic site amplification for central and eastern North America</article-title>,&#x201d; in <conf-name>15th World Conf. on Earthquake Engineering</conf-name>, <conf-loc>Lisbon, Portugal</conf-loc>, <conf-date>September 24 to September 28, 2012</conf-date>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krauss</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Do</surname>
<given-names>X. A.</given-names>
</name>
<name>
<surname>Huck</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Deep neural networks, gradient-boosted trees, random forests: statistical arbitrage on the S&#x26;P 500</article-title>. <source>Eur. J. Operational Res.</source> <volume>259</volume> (<issue>2</issue>), <fpage>689</fpage>&#x2013;<lpage>702</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejor.2016.10.031</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kwak</surname>
<given-names>D. Y.</given-names>
</name>
<name>
<surname>Brandenberg</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Mikami</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Prediction equations for estimating shear-wave velocity from combined geotechnical and geomorphic indexes based on Japanese data set</article-title>. <source>Bull. Seismol. Soc. Am.</source> <volume>105</volume> (<issue>4</issue>), <fpage>1919</fpage>&#x2013;<lpage>1930</lpage>. <pub-id pub-id-type="doi">10.1785/0120140326</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kwok</surname>
<given-names>O. L. A.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Kwak</surname>
<given-names>D. Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>P.-L.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Taiwan-specific model for V<sub>S30</sub> prediction considering between-proxy correlations</article-title>. <source>Earthq. Spectra</source> <volume>34</volume> (<issue>4</issue>), <fpage>1973</fpage>&#x2013;<lpage>1993</lpage>. <pub-id pub-id-type="doi">10.1193/061217EQS113M</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<collab>National Research Institute for Earth Science and Disaster Resilience</collab> (<year>2019</year>). <article-title>NIED K-NET, KiK-net, national research Institute for Earth science and disaster resilience</article-title>. <source>Natl. Res. Inst. Earth Sci. Disaster Resil.</source> <volume>2019</volume>. <pub-id pub-id-type="doi">10.17598/NIED.0004</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ohta</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Goto</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>1978</year>). <article-title>Empirical shear wave velocity equations in terms of characteristic soil indexes</article-title>. <source>Earthq. Eng. Struct. Dyn.</source> <volume>6</volume> (<issue>2</issue>), <fpage>167</fpage>&#x2013;<lpage>187</lpage>. <pub-id pub-id-type="doi">10.1002/eqe.4290060205</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Panza</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Agosta</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Rustichelli</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vinciguerra</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ougier-Simonin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dobbs</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Meso-to-microscale fracture porosity in tight limestones, results of an integrated field and laboratory study</article-title>. <source>Mar. Petroleum Geol.</source> <volume>103</volume>, <fpage>581</fpage>&#x2013;<lpage>595</lpage>. <pub-id pub-id-type="doi">10.1016/j.marpetgeo.2019.01.043</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pappalardo</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Correlation between P-wave velocity and physical&#x2013;mechanical properties of intensely jointed dolostones, Peloritani mounts, NE Sicily</article-title>. <source>Rock Mech. Rock Eng.</source> <volume>48</volume> (<issue>4</issue>), <fpage>1711</fpage>&#x2013;<lpage>1721</lpage>. <pub-id pub-id-type="doi">10.1007/s00603-014-0607-8</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parker</surname>
<given-names>G. A.</given-names>
</name>
<name>
<surname>Harmon</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Hashash</surname>
<given-names>Y. M.</given-names>
</name>
<name>
<surname>Kottke</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Rathje</surname>
<given-names>E. M.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Proxy&#x2010;based V<sub>S30</sub> estimation in central and eastern North America</article-title>. <source>Bull. Seismol. Soc. Am.</source> <volume>107</volume> (<issue>1</issue>), <fpage>117</fpage>&#x2013;<lpage>131</lpage>. <pub-id pub-id-type="doi">10.1785/0120160101</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paul</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chatterjee</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Prediction of compressional wave velocity using regression and neural network modeling and estimation of stress orientation in Bokaro Coalfield, India</article-title>. <source>Pure Appl. Geophys.</source> <volume>175</volume> (<issue>1</issue>), <fpage>375</fpage>&#x2013;<lpage>388</lpage>. <pub-id pub-id-type="doi">10.1007/s00024-017-1672-1</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Varoquaux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michel</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Thirion</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Grisel</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Scikit-learn: machine learning in Python</article-title>. <source>J. Mach. Learn. Res.</source> <volume>12</volume>, <fpage>2825</fpage>&#x2013;<lpage>2830</lpage>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pickett</surname>
<given-names>G. R.</given-names>
</name>
</person-group> (<year>1963</year>). <article-title>Acoustic character logs and their applications in formation evaluation</article-title>. <source>J. Petroleum Technol.</source> <volume>15</volume> (<issue>6</issue>), <fpage>659</fpage>&#x2013;<lpage>667</lpage>. <pub-id pub-id-type="doi">10.2118/452-PA</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rahimi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wood</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Wotherspoon</surname>
<given-names>L. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Influence of soil aging on SPT-Vs correlation and seismic site classification</article-title>. <source>Eng. Geol.</source> <volume>272</volume>, <fpage>105653</fpage>. <pub-id pub-id-type="doi">10.1016/j.enggeo.2020.105653</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rahman</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sarkar</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Lithological control on the estimation of uniaxial compressive strength by the P-wave velocity using supervised and unsupervised learning</article-title>. <source>Rock Mech. Rock Eng.</source> <volume>54</volume>, <fpage>3175</fpage>&#x2013;<lpage>3191</lpage>. <pub-id pub-id-type="doi">10.1007/s00603-021-02445-8</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roy</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kodikara</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Das</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Effect of water saturation on the fracture and mechanical properties of sedimentary rocks</article-title>. <source>Rock Mech. Rock Eng.</source> <volume>50</volume> (<issue>10</issue>), <fpage>2585</fpage>&#x2013;<lpage>2600</lpage>. <pub-id pub-id-type="doi">10.1007/s00603-017-1253-8</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Samui</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sitharam</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Support vector machine for evaluating seismic-liquefaction potential using shear wave velocity</article-title>. <source>J. Appl. Geophys.</source> <volume>73</volume> (<issue>1</issue>), <fpage>8</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1016/j.jappgeo.2010.10.005</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Seo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Machine-learning-based surface ground-motion prediction models for South Korea with low-to-moderate seismicity</article-title>. <source>Bull. Seismol. Soc. Am.</source> <volume>112</volume> (<issue>3</issue>), <fpage>1549</fpage>&#x2013;<lpage>1564</lpage>. <pub-id pub-id-type="doi">10.1785/0120210244</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Si</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Di</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Experimental study of water saturation effect on acoustic velocity of sandstones</article-title>. <source>J. Nat. Gas Sci. Eng.</source> <volume>33</volume>, <fpage>37</fpage>&#x2013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1016/j.jngse.2016.05.002</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sil</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Haloi</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Empirical correlations with standard penetration test (SPT)-N for estimating shear wave velocity applicable to any region</article-title>. <source>Int. J. Geosynth. Ground Eng.</source> <volume>3</volume> (<issue>3</issue>), <fpage>1</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1007/s40891-017-0099-1</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kanli</surname>
<given-names>A. I.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Estimating shear wave velocities in oil fields: a neural network approach</article-title>. <source>Geosciences J.</source> <volume>20</volume> (<issue>2</issue>), <fpage>221</fpage>&#x2013;<lpage>228</lpage>. <pub-id pub-id-type="doi">10.1007/s12303-015-0036-z</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sousa</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>del R&#xed;o</surname>
<given-names>L. M. S.</given-names>
</name>
<name>
<surname>Calleja</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>de Argandona</surname>
<given-names>V. G. R.</given-names>
</name>
<name>
<surname>Rey</surname>
<given-names>A. R.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Influence of microfractures and porosity on the physico-mechanical properties and weathering of ornamental granites</article-title>. <source>Eng. Geol.</source> <volume>77</volume> (<issue>1-2</issue>), <fpage>153</fpage>&#x2013;<lpage>168</lpage>. <pub-id pub-id-type="doi">10.1016/j.enggeo.2004.10.001</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>C.-G.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>C.-S.</given-names>
</name>
<name>
<surname>Son</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shin</surname>
<given-names>J. S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Correlations between shear wave velocity and <italic>in-situ</italic> penetration test results for Korean soil deposits</article-title>. <source>Pure Appl. Geophys.</source> <volume>170</volume> (<issue>3</issue>), <fpage>271</fpage>&#x2013;<lpage>281</lpage>. <pub-id pub-id-type="doi">10.1007/s00024-012-0516-2</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname>
<given-names>C.-C.</given-names>
</name>
<name>
<surname>Kishida</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kuo</surname>
<given-names>C.-H.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Unified correlation between SPT&#x2013;N and shear wave velocity for a wide range of soil types considering strain-dependent behavior</article-title>. <source>Soil Dyn. Earthq. Eng.</source> <volume>126</volume>, <fpage>105783</fpage>. <pub-id pub-id-type="doi">10.1016/j.soildyn.2019.105783</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>On a new method of estimating shear wave velocity from conventional well logs</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>180</volume>, <fpage>105</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2019.05.033</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Establishing region-specific N&#x2013;Vs relationships through hierarchical Bayesian modeling</article-title>. <source>Eng. Geol.</source> <volume>287</volume>, <fpage>106105</fpage>. <pub-id pub-id-type="doi">10.1016/j.enggeo.2021.106105</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yasar</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Erdogan</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Correlating sound velocity with the density, compressive strength and Young&#x27;s modulus of carbonate rocks</article-title>. <source>Int. J. Rock Mech. Min. Sci.</source> <volume>41</volume> (<issue>5</issue>), <fpage>871</fpage>&#x2013;<lpage>875</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijrmms.2004.01.012</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yousef</surname>
<given-names>W. A.</given-names>
</name>
<name>
<surname>Ibrahime</surname>
<given-names>O. M.</given-names>
</name>
<name>
<surname>Madbouly</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Mahmoud</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Learning meters of Arabic and English poems with recurrent neural networks: A step forward for language understanding and synthesis</source>. <comment>arXiv preprint arXiv:1905.05700</comment>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>H.-R.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.-Y.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Q.-Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Improvement of petrophysical workflow for shear wave velocity prediction based on machine learning methods for complex carbonate reservoirs</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>192</volume>, <fpage>107234</fpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2020.107234</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>