<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Built Environ.</journal-id>
<journal-title>Frontiers in Built Environment</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Built Environ.</abbrev-journal-title>
<issn pub-type="epub">2297-3362</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1343398</article-id>
<article-id pub-id-type="doi">10.3389/fbuil.2024.1343398</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Built Environment</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Efficient data-driven machine learning models for scour depth predictions at sloping sea defences</article-title>
<alt-title alt-title-type="left-running-head">Habib et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fbuil.2024.1343398">10.3389/fbuil.2024.1343398</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Habib</surname>
<given-names>M. A.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2585801/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1323253/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>O&#x2019;Sullivan</surname>
<given-names>John. J.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1322046/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/958278/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>UCD School of Civil Engineering</institution>, <institution>UCD Dooge Centre for Water Resources Research and UCD Earth Institute</institution>, <institution>University College Dublin</institution>, <addr-line>Dublin</addr-line>, <country>Ireland</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Engineering</institution>, <institution>University of Warwick</institution>, <addr-line>Coventry</addr-line>, <country>United Kingdom</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/256631/overview">Nobuhito Mori</ext-link>, Kyoto University, Japan</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1203875/overview">Sooyoul Kim</ext-link>, Kumamoto University, Japan</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1867374/overview">Mohammad Khajehzadeh</ext-link>, Islamic Azad University, Anar, Iran</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: M. Salauddin, <email>md.salauddin@ucd.ie</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>02</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>10</volume>
<elocation-id>1343398</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>11</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>01</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Habib, Abolfathi, O&#x2019;Sullivan and Salauddin.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Habib, Abolfathi, O&#x2019;Sullivan and Salauddin</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Seawalls are critical defence infrastructures in coastal zones that protect hinterland areas from storm surges, wave overtopping and soil erosion hazards. Scouring at the toe of sea defences, caused by wave-induced accretion and erosion of bed material imposes a significant threat to the structural integrity of coastal infrastructures. Accurate prediction of scour depths is essential for appropriate and efficient design and maintenance of coastal structures, which serve to mitigate risks of structural failure through toe scouring. However, limited guidance and predictive tools are available for estimating toe scouring at sloping structures. In recent years, Artificial Intelligence and Machine Learning (ML) algorithms have gained interest, and although they underpin robust predictive models for many coastal engineering applications, such models have yet to be applied to scour prediction. Here we develop and present ML-based models for predicting toe scour depths at sloping seawall. Four ML algorithms, namely, Random Forest (RF), Gradient Boosted Decision Trees (GBDT), Artificial Neural Networks (ANNs), and Support Vector Machine Regression (SVMR) are utilised. Comprehensive physical modelling measurement data is utilised to develop and validate the predictive models. A Novel framework for feature selection, feature importance, and hyperparameter tuning algorithms are adopted for pre- and post-processing steps of ML-based models. In-depth statistical analyses are proposed to evaluate the predictive performance of the proposed models. The results indicate a minimum of 80% prediction accuracy across all the algorithms tested in this study and overall, the SVMR produced the most accurate predictions with a Coefficient of Determination (<italic>r</italic>
<sup>2</sup>) of 0.74 and a Mean Absolute Error (MAE) value of 0.17. The SVMR algorithm also offered most computationally efficient performance among the algorithms tested. The methodological framework proposed in this study can be applied to scouring datasets for rapid assessment of scour at coastal defence structures, facilitating model-informed decision-making.</p>
</abstract>
<kwd-group>
<kwd>random forest</kwd>
<kwd>gradient boosted decision trees</kwd>
<kwd>Support Vector Machine Regression</kwd>
<kwd>marine and coastal management</kwd>
<kwd>coastal hazards mitigation</kwd>
<kwd>toe scouring</kwd>
<kwd>sloping structures</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Coastal and Offshore Engineering</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Scouring is the process of gradual erosion and removal of bed materials in the vicinity of coastal structures caused by hydrodynamic forces from waves and tidal currents. In addition to the hydrodynamic forces from tides and waves, which can be compounded by climate change influences, critical infrastructures including underwater pipelines, coastal defence structures, and coastal zone management processes such as dredging can contribute to conditions that are favourable to increased seabed scouring through the disruption of natural sediment transport processes and the alteration of the prevailing hydrodynamic environment in the nearshore region. Scouring at the toes of critical coastal defence structures (e.g., sloping and vertical seawalls) can result in the loss of structural integrity (<xref ref-type="bibr" rid="B45">Salauddin and Pearson, 2019a</xref>; <xref ref-type="bibr" rid="B46">Salauddin and Pearson, 2019b</xref>; <xref ref-type="bibr" rid="B53">Tseng et al., 2022</xref>) and ultimate failure, and is particularly critical in the management of coastal flood risks. Toe scouring can elevate wave overtopping discharge at defences, by increasing water depth at the defence and causing the formation of larger waves at the structure (<xref ref-type="bibr" rid="B36">Peng et al., 2023</xref>). The sedimentation and scouring in the vicinity of coastal structures can alter the bottom topography and bed slope, which in turn can influence wave shoaling and breaking processes and alters the turbulent kinetic energy budget of waves and their potential to overtop defences (<xref ref-type="bibr" rid="B36">Peng et al., 2023</xref>). Given that extreme events in coastal regions are predicted to increase in intensity and frequency under climate change scenarios, increased exposure to toe scouring at coastal defences is likely to be an increasing issue in the coming years (<xref ref-type="bibr" rid="B11">Fitri et al., 2019</xref>; <xref ref-type="bibr" rid="B43">Salauddin and O&#x2019;Sullivan, 2022</xref>). The availability of accurate methods to predict toe scour depths is, therefore, critical for mitigating scour related risks.</p>
<p>Reliable prediction of scour depths at coastal defences challenging and is influenced by complex wave-structure interactions and a range of nearshore processes (hydrodynamic and morphological). The prediction of scour depth, therefore, involves the consideration of parameters that reflect the diverse processes. These relate to wave and current conditions, tide and wave approach angles, sediment and bathymetric characteristics and features, and water depth at the structure (<xref ref-type="bibr" rid="B32">M&#xfc;ller et al., 2008</xref>; <xref ref-type="bibr" rid="B38">Pourzangbar et al., 2017a</xref>). For example, the scouring patterns observed in fine and coarse grained bed material are distinctly different (<xref ref-type="bibr" rid="B37">Pourzangbar et al., 2017b</xref>). Previous studies also highlighted that scour depth from regular waves are generally larger than those observed for irregular waves.</p>
<p>A significant number of beaches globally are coarse-gained shingle beaches, often with man-made coastal defences such as vertical seawalls or sloping structures (<xref ref-type="bibr" rid="B60">Powell and Lowe 1994</xref>; <xref ref-type="bibr" rid="B44">Salauddin and Pearson, 2018</xref>; <xref ref-type="bibr" rid="B63">Salauddin and Pearson, 2020</xref>). Although the literature (e.g., <xref ref-type="bibr" rid="B38">Pourzangbar et al., 2017a</xref>; <xref ref-type="bibr" rid="B37">Pourzangbar et al., 2017b</xref>) has demonstrated the robust performance of ML algorithms in predicting scour depth at sandy beaches, the capabilities of ML techniques for predicting scour in shingle foreshores are much less reported. The recent study by <xref ref-type="bibr" rid="B47">Salauddin et al. (2023)</xref> focussed on evaluating the effectiveness of ML algorithms for predicting scour depths at vertical seawalls and showed that ML models were able to predict scour depths with good accuracy for experimental data. Nevertheless, there remains a scope of the application of such algorithms to other structure types such as a sloping structure on a permeable shingle bed and investigate the performance of such algorithms in predicting scour depths for the same.</p>
<p>Here we present for the first time the development and testing of ML algorithms (namely, Support Vector Machines Regression (SVMR), Gradient Boosted Decision Trees (GBDT), Random Forests (RF) and Artificial Neural Networks (ANN)) at a sloping structure with a sloping shingle foreshore. The models were trained and tested on a physical modelling experimental dataset of scour depths at a 1 in 2 (1&#xa0;V:2H) impermeable sloping seawall located on a permeable 1 in 20 (1&#xa0;V:20H) shingle foreshore. Advanced novel pre-processing and post-processing techniques such as feature selection and feature importance are proposed to facilitate ML-based modelling for scouring datasets and we devise a stepwise methodological framework for scouring prediction. The predictive performance of ML models are investigated through well-established statistical metrics. The key objectives of this study are (i) to develop a robust methodological framework to use data driven ML algorithms for predicting scour depth at coastal defences, and (ii) quantify the predictive performance of selected ML-based models for estimating scour depths at sloping coastal sea defences.</p>
</sec>
<sec id="s2">
<title>2 Scour prediction methods</title>
<p>Existing studies assessing scour at sea defences such as vertical seawalls and sloping seawalls are typically underpinned by numerical, laboratory and field-based modelling approaches to derive empirical relations and engineering guidance. <xref ref-type="bibr" rid="B13">Fowler (1992)</xref> developed empirical formulae for toe scour depth based on physical modelling of scouring at a vertical seawall placed on a sandy foreshore. <xref ref-type="bibr" rid="B62">Wallis et al. (2010)</xref> and <xref ref-type="bibr" rid="B48">Sutherland et al. (2003</xref>, <xref ref-type="bibr" rid="B50">2006)</xref> proposed an improved guidance for predicting scour depths at vertical walls constructed on sandy foreshores using field and laboratory observations. These authors also claimed that for the tested conditions, maximum scour depths at a plain vertical wall were similar to those observed for a 1 in 2 sloping seawall. In recent years, <xref ref-type="bibr" rid="B45">Salauddin and Pearson (2019a)</xref>, <xref ref-type="bibr" rid="B46">Salauddin and Pearson, (2019b)</xref> conducted a comprehensive suite of laboratory-based physical modelling experiments to characterise scouring at both vertical and sloping structures on shingle foreshores, subjected to a wide range of irregular wave conditions (including storm and swell sea states).</p>
<p>The review of literature relating to scour at seawalls reveals a substantial correlation between toe scour depth and relative water depth at the toe (<italic>h</italic>
<sub>
<italic>t</italic>
</sub>
<italic>/L</italic>
<sub>
<italic>0m</italic>
</sub>), where, <italic>h</italic>
<sub>
<italic>t</italic>
</sub> is the toe water depth (m) and <italic>L</italic>
<sub>
<italic>0m</italic>
</sub> is the mean deep water wavelength (m), for defences on sandy foreshores. <xref ref-type="bibr" rid="B49">Sutherland et al. (2008)</xref> proposed an empirical relationship (Eq. <xref ref-type="disp-formula" rid="e1">1</xref>) between the dimensionless scour depth (<italic>S</italic>
<sub>
<italic>t</italic>
</sub>
<italic>/H</italic>
<sub>
<italic>s</italic>
</sub>), [calculated from scour depth <italic>S</italic>
<sub>
<italic>t</italic>
</sub> (m) and significant wave height (m), <italic>H</italic>
<sub>
<italic>s</italic>
</sub> (m)], and relative toe water depth (<italic>h</italic>
<sub>
<italic>t</italic>
</sub>
<italic>/L</italic>
<sub>
<italic>0m</italic>
</sub>) for the prediction of toe scour depth at a plain vertical seawall in a sandy beach. This was later verified by <xref ref-type="bibr" rid="B32">M&#xfc;ller et al. (2008)</xref>. Similar findings were also observed for scouring at a plain vertical wall with a shingle foreshore slope (<xref ref-type="bibr" rid="B45">Salauddin and Pearson, 2019a</xref>; <xref ref-type="bibr" rid="B46">Salauddin and Pearson, 2019b</xref>). <xref ref-type="bibr" rid="B49">Sutherland et al. (2008)</xref> also proposed an empirically based equation to predict the toe scour depth for vertical seawalls considering the influence of beach slope (Eq. <xref ref-type="disp-formula" rid="e2">2</xref>).<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4.5</mml:mn>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>8</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>0.01</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>0.01</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>6.8</mml:mn>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0.207</mml:mn>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1.51</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>5.8</mml:mn>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>0.137</mml:mn>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where, <italic>S</italic>
<sub>
<italic>t</italic>
</sub> and <inline-formula id="inf1">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the toe scour depth and maximum toe scour depth, respectively, <inline-formula id="inf2">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is significant wave height (&#x3d;highest one-third of wave heights), <inline-formula id="inf3">
<mml:math id="m5">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is beach slope, <inline-formula id="inf4">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is toe water depth, <inline-formula id="inf5">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is deep-water wavelength based on <italic>T</italic>
<sub>
<italic>m</italic>
</sub>, where <inline-formula id="inf6">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the mean wave period.</p>
<p>Numerical modelling tools have also developed and applied to simulate scour behaviour at coastal defences (<xref ref-type="bibr" rid="B35">Peng et al., 2018</xref>; <xref ref-type="bibr" rid="B36">Peng et al., 2023</xref>; <xref ref-type="bibr" rid="B58">Yeganeh-Bakhtiary et al., 2020</xref>). For example, <xref ref-type="bibr" rid="B36">Peng et al. (2023)</xref> utilized Reynolds Averaged Navier&#x2013;Stokes equations (RANS) and the Volume of Fluid (VOF) modelling technique, coupled with wave-sediment transport and morphological factors, to simulate scour dynamics in front of an impermeable plane vertical seawall under specific wave conditions. However, robust numerical modelling techniques for estimating scour in wave environments still remain limited, largely as a result of the complexity of multiphase flow simulations, but also as a result of the high computational requirements (due to the involvement of intrinsic equations) that are involved. For example, in numerical simulations of estimating scour depths, uncertainty is induced from the dependency of such models on empirical parameters of the scouring process (<xref ref-type="bibr" rid="B55">Yang et al., 2018</xref>).</p>
<p>In recent years, with advancements in data science and computational resources, Artificial Intelligence (AI) in the form of Machine Learning (ML) has been successfully employed to address a wide range of coastal engineering problems. For example, significant research relating to the development of AI based decision-support algorithms for the prediction of wave characteristics (<xref ref-type="bibr" rid="B56">Yeganeh-Bakhtiary et al., 2023</xref>) and wave overtopping at coastal defences has been undertaken (see, for example, <xref ref-type="bibr" rid="B4">den Bieman et al., 2021a</xref>, 2021b; <xref ref-type="bibr" rid="B5">den Bieman et al., 2020</xref>; <xref ref-type="bibr" rid="B9">Elbisy, 2023</xref>; <xref ref-type="bibr" rid="B10">Elbisy and Elbisy, 2021</xref>; <xref ref-type="bibr" rid="B19">Habib et al., 2022b</xref>; <xref ref-type="bibr" rid="B18">Habib et al., 2023a</xref>; <xref ref-type="bibr" rid="B16">Habib et al., 2023b</xref>). <xref ref-type="bibr" rid="B17">Habib et al. (2022a)</xref> has provided an overview of recent studies on the applications of ML approaches in coastal engineering problems.</p>
<p>Data-driven ML modelling approaches have been applied to predict scour depths at vertical breakwaters. <xref ref-type="bibr" rid="B38">Pourzangbar et al. (2017a)</xref>, <xref ref-type="bibr" rid="B37">Pourzangbar et al. (2017b)</xref> successfully applied several ML algorithms, including Genetic Programming (GP), Artificial Neural Network (ANN), Support Vector Machine Regression (SVMR) and the M5&#x2019; Decision Tree model to predict scour depth from physical modelling data for impermeable vertical breakwaters with sandy foreshores. However, the development to date of ML-based scour prediction models have thus far been applied to vertical breakwaters and sandy foreshores with fine grains. Previous studies however have not dealt with the prediction of scour depth at a sloping structure on a permeable shingle foreshore using advanced ML algorithms, which has been addressed for the first time in this work.</p>
</sec>
<sec sec-type="materials|methods" id="s3">
<title>3 Materials and methods</title>
<sec id="s3-1">
<title>3.1 Scouring dataset</title>
<p>The scour dataset used in this study was obtained from experimental studies conducted in a 2D wave flume, 22&#xa0;m long, 0.6&#xa0;m wide, and 1&#xa0;m deep (<xref ref-type="fig" rid="F1">Figure 1</xref>), at the University of Warwick&#x2019;s Water Engineering Laboratory (<xref ref-type="bibr" rid="B46">Salauddin and Pearson, 2019b</xref>). The flume was equipped with a piston-type wave paddle, six Wave Gauges (WG) and active adsorption system capable of generating monochromatic and random waves, generating realistic sea states in the wave channel. The dataset consisted of over 120 experiments in which the scour characteristics at the toe of a sloping wall (1:2) with a shingle foreshore, of approximately 6&#xa0;m length, on a 1:20 slope were observed and included a comprehensive range of incident wave conditions including both impulsive and non-impulsive waves. The JONSWAP wave spectrum with a peak-enhancement factor of 3.3 was applied to generate incident waves that were representative of the young sea state. The relative crest freeboard (<italic>R</italic>
<sub>
<italic>c</italic>
</sub>
<italic>/H</italic>
<sub>
<italic>m0</italic>
</sub>), (where <italic>Rc</italic> is the crest-freeboard of the defence structure and <italic>H</italic>
<sub>
<italic>mo</italic>
</sub> is the wave height at the toe of the structure) ranged from 0.5 to 5.0 and this was achieved by applying six different types of toe water depths. The scouring characteristics were measured for both impulsive and non-impulsive wave conditions. The dataset comprising of 120 sets of observations, was split into a train-test set of 70%&#x2013;30%.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Schematic of the Experimental setup for measuring scour depth at a sloping wall with a shingle foreshore (Adopted from <xref ref-type="bibr" rid="B46">Salauddin and Pearson, 2019b</xref>).</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g001.tif"/>
</fig>
<p>For each test configuration, the scour depth was measured at the toe of the structure and at different locations along the wave flume in front of the structure. The maximum scour depth was then determined from these measurements. Analysis of the experimental data showed that, for the wave conditions tested, the maximum scour depth occurred at the toe of the structure. An insight into the database in terms of statistical correlation (Pearson R) revealed very low correlative relations between the scour parameters (described in the Glossary section) and the relative scour depth (&#x3d;<italic>S</italic>
<sub>
<italic>t</italic>
</sub>
<italic>/H</italic>
<sub>
<italic>m0</italic>
</sub>; where, <italic>S</italic>
<sub>
<italic>t</italic>
</sub> is the measured scour depth and <italic>H</italic>
<sub>
<italic>m0</italic>
</sub> is the water depth at the toe of the structure). No negative correlation was observed between the variables, however, only <italic>R</italic>
<sub>
<italic>c</italic>
</sub>
<italic>/H</italic>
<sub>
<italic>1/3,deep</italic>
</sub> and <italic>I</italic>
<sub>
<italic>r</italic>
</sub> showed a maximum correlation of 0.25 with the relative scour depth. Two kernel (ANN and SVMR) and two DT-based (RF and GBDT) algorithms were investigated in the study of <xref ref-type="bibr" rid="B18">Habib et al. (2023a)</xref> and it was reported that the algorithms performed satisfactorily in predicting wave overtopping at a vertical sea wall. The algorithms are hence also investigated for a scour dataset, since the intrinsic nature of the scour dataset is similar to what was applied in the overtopping study (<xref ref-type="bibr" rid="B16">Habib et al., 2023b</xref>).</p>
<p>The workflow followed in the data preparation, together with model development and testing is summarised in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The methodological approach adopted for the ML based modelling.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g002.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 SVMR</title>
<p>SVMR is a category of supervised ML algorithms and an extension of the classification-based Support Vector Machines (SVM), typically employed for regression tasks (<xref ref-type="bibr" rid="B33">Noori et al., 2022</xref>). SVMR algorithms aim to minimize the prediction error and simultaneously maximize the margin around the fitting function, effectively identifying the best-fit function for a given dataset. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates a typical workflow structure for SVMR. For a regression problem using a training dataset containing interlinked input features (<italic>x</italic>
<sub>
<italic>i</italic>
</sub>) and target values (<italic>y</italic>
<sub>
<italic>i</italic>
</sub>), SVMR deduces a function <italic>f</italic>(<italic>x</italic>) that predicts the target values <italic>y</italic> based on input features <italic>x</italic>. The fundamental goal of SVMR is to construct a hyperplane that closely fits the training data within a specified tolerance of error margin (<italic>&#x3b5;</italic>). Feature points inside the epsilon tube surrounding the hyperplane are regarded as support vectors, <italic>w,</italic> and do not incur any penalties. Points outside of this tube are penalized because they add to the error. The loss function is determined using Eq. <xref ref-type="disp-formula" rid="e3">3</xref>:<disp-formula id="e3">
<mml:math id="m9">
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b5;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where, <inline-formula id="inf7">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf8">
<mml:math id="m11">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b5;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are slack variables that gauge how far the outliers are from the <italic>&#x3b5;</italic>-tube, <italic>N</italic> is the total number of slack variables and <italic>C</italic> is a regularization factor that can be adjusted to determine the flatness of the hyperplane.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The workflow of a SVMR algorithm adopted in this study.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g003.tif"/>
</fig>
<p>The main goal of the optimisation approaches related to SVMR is deducing the optimal values for <italic>w</italic>, and the slack variables, <inline-formula id="inf9">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf10">
<mml:math id="m13">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b5;</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> SVMR. The objective is to minimize the regularization term while ensuring that errors are within &#x3b5; and slack variables remain non-negative. If non-linearity exists, the feature data is projected onto kernel space, a higher-dimensional hyperplane, which improves the model&#x2019;s accuracy. The function <inline-formula id="inf11">
<mml:math id="m14">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> defines the kernel space, and in this study, a Gaussian Radial Basis kernel function (RBF, Eq. <xref ref-type="disp-formula" rid="e4">4</xref>) is utilised:<disp-formula id="e4">
<mml:math id="m15">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="italic">exp</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where, <inline-formula id="inf12">
<mml:math id="m16">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the kernel parameter. Gaussian RBF kernel is suitable for datasets with unknown or challenging-to-trace intrinsic feature characteristics (<xref ref-type="bibr" rid="B41">Roushangar and Koosheh, 2015</xref>). This is because the RBF kernel, based on the Taylor Series expansion, can accommodate an infinite number of feature dimensions. SVMR is particularly known for producing robust predictions when dealing with non-linear and high dimensional data (e.g., <xref ref-type="bibr" rid="B22">Kawashima and Kumano, 2017</xref>; <xref ref-type="bibr" rid="B26">Lan et al., 2023</xref>), similar to the dataset used in this study.</p>
</sec>
<sec id="s3-3">
<title>3.3 ANN</title>
<p>ANNs are well established in coastal engineering applications for tackling classification and regression tasks by mapping inputs to outputs, assigning weights to specific inputs, estimating and minimizing a loss function (example.g., <xref ref-type="bibr" rid="B39">Raikar et al., 2018</xref>; <xref ref-type="bibr" rid="B54">Verhaeghe et al., 2008</xref>; <xref ref-type="bibr" rid="B59">Zanuttigh et al., 2016</xref>; <xref ref-type="bibr" rid="B12">Formentin et al., 2017</xref>; <xref ref-type="bibr" rid="B61">EurOtop, 2018</xref>; <xref ref-type="bibr" rid="B18">Habib et al., 2023a</xref>; <xref ref-type="bibr" rid="B16">Habib et al., 2023b</xref>). <xref ref-type="fig" rid="F4">Figure 4</xref> illustrates the workflow of a feed-forward and back propagation ANN algorithm, including the input, hidden, and output layers. The input layer receives data from the training set. The information is only communicated to and from each layer within the neural network and not between neurones in the same layer. The model&#x2019;s hidden layers are responsible for assigning numerical weights to the incoming information from the input layers and to the activation functions. The output layer of the network estimates the quantity predicted by the activation functions and then calculates the dependent feature(s) from the independent feature(s) in the input layers (<xref ref-type="bibr" rid="B1">Babaee et al., 2021</xref>; <xref ref-type="bibr" rid="B23">Khosravi et al., 2023</xref>).</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Schematic of a feed-forward and back propagation ANN algorithm adopted in this study.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g004.tif"/>
</fig>
<p>A Multi-Layered Perceptron (MLP) ANN, which is feed-forward and back-propagation in nature, is adopted in this study (<xref ref-type="fig" rid="F4">Figure 4</xref>). The term feed-forward and back-propagation essentially means that until a predetermined allowable error rate is achieved, the error rates are minimized by altering the loss functions through a combination of feed-forward (exchange of information from Input to Hidden to Output Layers) and back propagation (exchange of information from Output to Hidden and Hidden to Input Layers). The adjustment of weights and biases during the backpropagation stage is determined by the error rate. This process involves assigning new weights and activation functions to the hidden layers. The optimization of the number of hidden layers is usually based on the complexity of the input data, aiming to minimize prediction error (<xref ref-type="bibr" rid="B7">Elbeltagi et al., 2021</xref>; <xref ref-type="bibr" rid="B8">2022</xref>).</p>
</sec>
<sec id="s3-4">
<title>3.4 RF and GBDT</title>
<p>RF and GBDT algorithms are typically categorised as Decision Trees (DTs). DTs are supervised machine learning algorithms used to predict an output variable (i.e., dependent or target variable) based on a set of independent variables (i.e., features). DTs are capable of tackling both classification and regression problems. In regression, they predict continuous or numerical output variables, while in classification, they predict class labels for discrete output variables (<xref ref-type="bibr" rid="B57">Yeganeh-Bakhtiary et al., 2022</xref>).</p>
<p>In the case of regression-based DTs, the training data is iteratively partitioned into rectangular regions, and the mean and median values within each region is estimated until a pre-determined stopping criteria are met. For example, given a training dataset, <italic>X</italic> &#x3d; <inline-formula id="inf13">
<mml:math id="m17">
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf14">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents an input feature vector for the <italic>i</italic>th training dataset and <inline-formula id="inf15">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the corresponding output, the DT algorithm divides <italic>X</italic> into a series of rectangular regions, denoted as R1, R2, R3, etc. For each region, the median and mean are estimated to serve as the prediction value <italic>P</italic> for that corresponding region. The final DT is constructed using the input features that distinctly divide these rectangular sections and yield the output variable with the smallest variance. DTs are commonly used in prediction tasks due to their ability to handle noise and non-linearity in input data independently (<xref ref-type="bibr" rid="B34">Pedregosa et al., 2011</xref>; <xref ref-type="bibr" rid="B25">Kotu and Deshpande, 2015</xref>; <xref ref-type="bibr" rid="B56">Yeganeh-Bakhtiary et al., 2023</xref>).</p>
<p>A Random Forest (RF) algorithm is an ensemble of DTs constructed from a random sub-set of training data. <xref ref-type="fig" rid="F5">Figure 5</xref> illustrates a schematic of methodological workflow for RF modelling approach.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>The workflow of a RF algorithm [Adopted from <xref ref-type="bibr" rid="B18">Habib et al. (2023a)</xref>].</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g005.tif"/>
</fig>
<p>RF model aims to reduce overfitting and enhance generalization by minimizing overexposure to any specific set of training data. The final prediction from tRF is the average of predictions made by individual DTs, often referred to as bagging. An additional advantage of RF is its capability handle both categorical and numerical data, further minimising overfitting.</p>
<p>The boosting strategy is another method for enhancing DTs&#x2019; predictive capabilities. An example of a Boosting approach is the GBDT algorithm (see <xref ref-type="fig" rid="F6">Figure 6</xref>). The Mean Squared Error (MSE) between the predicted and actual values is measured in the boosting technique using a loss function. During training, the boosting algorithm aims to minimize this loss function by assigning numerical coefficients to input data, often through gradient descent. The GBDT algorithm, in particular, is known for rapidly minimizing the loss function, resulting in faster and more accurate predictions from DT models (<xref ref-type="bibr" rid="B51">Sutton, 2005</xref>).</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The workflow of a GBDT algorithm. Adopted from <xref ref-type="bibr" rid="B19">Habib et al. (2023b)</xref>.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g006.tif"/>
</fig>
</sec>
<sec id="s3-5">
<title>3.5 Model optimization</title>
<sec id="s3-5-1">
<title>3.5.1 Hyperparameter tuning</title>
<p>Hyperparameters refer to the parameters of a ML algorithm that can be adjusted or tuned by the user, as opposed to model parameters, such as the coefficients of mapping functions, which are not user-accessible. Hyperparameter tuning is a crucial process for reducing overfitting and ensuring that the ML algorithm is well-suited for a specific set of input data. Hyperparameter tuning was conducted for all the ML adopted models in this study using the open-source <italic>scikit-learn library</italic> in Python (<xref ref-type="bibr" rid="B34">Pedregosa et al., 2011</xref>). <xref ref-type="table" rid="T1">Table 1</xref> summarises the optimum hyperparameters adopted for the SVMR, RF, GBDT, and ANN models. The SVMR algorithm&#x2019;s regularization parameter is represented by the C term in <xref ref-type="table" rid="T1">Table 1</xref>. The algorithm&#x2019;s &#x201c;engine&#x201d; is a function called the kernel that maps input parameters (independent variables) onto output values (dependent variable). This study investigates the performance of linear, polynomial and RBF kernels. Gamma (<xref ref-type="table" rid="T1">Table 1</xref>) is a kernel function coefficient. This study combines &#x201c;RandomizedSearch&#x201d; and a k-fold Cross Validation (CV) to find the best parameters. CV is a popular resampling method that eliminates bias from prediction models (<xref ref-type="bibr" rid="B34">Pedregosa et al., 2011</xref>; <xref ref-type="bibr" rid="B47">Salauddin et al., 2023</xref>). The data is randomly divided into k sets of nearly similar size for k-fold cross validation. The ML algorithms are first tested on these folds to validate the training, and then applied to the test set. The validation step ensures that the algorithms explicitly capture the variations and patterns in the training set. The function RandomizedSearchCV (RS) uses a set number of random combinations of hyperparameters. The RS function is particularly suitable for performing hypertuning when there are a large number of hyperparameters involved, i.e., similar to this work.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Set of hyperparameters and their optimised values.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Algorithm</th>
<th align="center">Source code of hyperparameters</th>
<th align="center">Best values</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">SVMR</td>
<td align="center">parameters &#x3d; {&#x27;kernel&#x27;: (&#x27;linear&#x27;, &#x27;rbf&#x27;,&#x27;poly&#x27;), &#x27;C&#x27;: [2, 5, 10, 15, 20],&#x27;gamma&#x27;: [&#x27;auto&#x27;,&#x27;scale&#x27;],&#x27;epsilon&#x27;: [0.1,0.2,0.5,0.3]}</td>
<td align="center">&#x2018;kernel&#x2019;: &#x2018;rbf; &#x2018;gamma&#x2019;: &#x2018;scale&#x2019;; &#x2018;epsilon&#x2019;: 0.1; &#x2018;C&#x2019;: 15</td>
</tr>
<tr>
<td align="center">RF</td>
<td align="center">param_grid &#x3d; { &#x27;n_estimators&#x27;: np.arange (50, 500, 10), &#x27;max_depth&#x27;: np.arange (3,12,3), &#x27;min_samples_split&#x27;: np.arange (2, 3), &#x27;min_samples_leaf&#x27;: np.arange (2, 3), &#x2018;learning_rate&#x2019;:np.arrange (0.01, 0.1, 0.01) }</td>
<td align="center">n_estimators&#x27;: 50, &#x2018;min_samples_split&#x2019;: 2, &#x2018;min_samples_leaf&#x2019;: 2, &#x27;max_depth&#x27;: 3, &#x2018;learning_rate&#x2019;: 0.06</td>
</tr>
<tr>
<td align="center">GBDT</td>
<td align="center">param_grid &#x3d; { &#x27;n_estimators&#x27;: np.arange (50, 100, 10), &#x27;max_depth&#x27;: np.arange (3, 12), &#x27;learning_rate&#x27;: np.arange (0.01, 0.1, 0.05), &#x27;colsample_bytree&#x27;: np.arange (0.5, 1.0, 0.1)}</td>
<td align="center">&#x2018;n_estimators&#x2019;: 70; &#x2018;max_depth&#x2019;: 10; &#x2018;learning_rate&#x2019;: 0.06; &#x2018;colsample_bytree&#x2019;: 0.5</td>
</tr>
<tr>
<td align="center">ANN</td>
<td align="center">param_grid &#x3d; { &#x27;hidden_layer_sizes&#x27;: [(150,100,50), (120,80,40), (100,50,30)], &#x27;max_iter&#x27;: np.arange (50,500,50), &#x27;activation&#x27;: [&#x27;logistic&#x27;,&#x27;tanh&#x27;, &#x27;relu&#x27;], &#x27;solver&#x27;: [&#x27;sgd&#x27;, &#x27;adam&#x27;], &#x27;alpha&#x27;: [0.0001, 0.0005], &#x27;learning_rate&#x27;: [&#x27;constant&#x27;,&#x27;adaptive&#x27;], }</td>
<td align="center">&#x2018;activation&#x2019;: &#x2018;relu&#x2019;; &#x2018;hidden layer sizes&#x2019;: [100,50,30]; &#x2018;max_iter&#x2019;: 150; &#x2018;alpha&#x2019;: 0.0001; &#x2018;solver&#x2019;: &#x2018;adam&#x2019;</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The key functional components of a DT network can be found in the hyperparameters of the RF model (<xref ref-type="table" rid="T1">Table 1</xref>). &#x201c;n_estimators&#x201d; determines the number of trees in an RF, while &#x201c;max_depth&#x201d; and &#x201c;min_samples_split&#x201d; help mitigating overfitting. In this study, a random search with Cross Validation (CV) was used for hyperparameter tuning in the RF model.</p>
<p>GBDT and RF are both based on DTs, with GBDT relying on gradient boosting. GBDT&#x2019;s hyperparameters (<xref ref-type="table" rid="T1">Table 1</xref>) determine the size of the decision tree that best suits the input data. &#x201c;learning_rate&#x201d; is crucial for reducing overfitting as it computes the weights of input features to converge the error in the loss function. &#x201c;max_depth&#x201d; also plays a role in reducing overfitting by limiting the number of nodes in the trees. For GBDT, hyperparameter tuning was performed using a random search with 5-fold cross validation. The scope of hyperparameter tuning with ANN is limited (<xref ref-type="bibr" rid="B20">Huang et al., 2012</xref>; <xref ref-type="bibr" rid="B14">Ghiasi et al., 2022</xref>). RS with a k-fold CV approach is implemented in this study to enhance the learning rate &#x201c;alpha,&#x201d; and the best model is determined based on the model loss criterion. Typical hyperparameter tuning values for the ANN models are adopted from <xref ref-type="bibr" rid="B27">LeCun et al. (2015)</xref> and <xref ref-type="bibr" rid="B15">Glorot and Bengio (2010)</xref>. The kernel function of the ANN is located in the hidden layers, and the user can predetermine both the number of layers and neurons in each layer. Additional to the &#x2018;alpha&#x2019; parameter&#x2019; and the activation function, the number of epochs was adjusted to attain the optimal set of hyperparameters for the ANN in this study.</p>
</sec>
<sec id="s3-5-2">
<title>3.5.2 Feature selection and feature transformation</title>
<p>Robust ML-based predictions can be challenging when dealing with high-dimensional data that can reduce the effectiveness and accuracy of machine learning algorithms due to data redundancy. Additionally, computational resource costs can increase due to prolonged algorithm runtimes. To address the issue of data redundancy, feature selection techniques are employed. These techniques aim to filter a subset of relevant features from a large dataset, effectively eliminating redundancy and irrelevance (<xref ref-type="bibr" rid="B2">Cai et al., 2018</xref>). Feature selection is typically achieved through statistics-based permutation combinations, which measure the correlation of individual features with a target feature. The most important features are then deduced based on their correlation scores, as highlighted by <xref ref-type="bibr" rid="B29">Liu and Motoda (2012)</xref> and <xref ref-type="bibr" rid="B6">Donnelly et al. (2024)</xref>.</p>
<p>Feature transformation is a technique used for extracting useful features from a large dataset, where the initial number of features is transformed into a new, more compact dataset with fewer but relevant features, while conserving the implicit and/or explicit information of the original dataset. One well-known feature transformation technique is Principal Component Analysis (PCA) (<xref ref-type="bibr" rid="B40">Roessner et al., 2011</xref>; <xref ref-type="bibr" rid="B33">Noori et al., 2022</xref>). PCA is particularly useful for capturing and reducing variance in large datasets by selecting the most relevant features that account for the majority of variance across the dataset. It is characterized as a dimensionality reduction technique that converts the original variables into uncorrelated principal components.</p>
<p>This study adopts a combination of feature selection and feature transformation techniques to discover and filter the most relevant features in the scour dataset. A Forward Sequential Feature Selection (FSFS) method is employed for feature selection. FSFS is a &#x201c;greedy&#x201d; method that iteratively builds a set of selected features (<italic>S</italic>) by adding new features, one at a time, and performing prediction tasks using a chosen estimator. In more concrete terms, FSFS starts with zero features and identifies the feature that, when used to train an estimator (e.g., linear regression in this study), maximizes a Cross Validation (CV) score. This process is repeated, adding one feature at a time, until all features in the dataset have been considered. The number of features that maximizes the CV score is considered the optimal number. FSFS is widely accepted for its simplicity and accuracy in estimating the number of important features in a dataset (<xref ref-type="bibr" rid="B30">Marcano-Cedeno et al., 2010</xref>). In this study, FSFS determined 10 parameters as the optimum number of features (see <xref ref-type="fig" rid="F7">Figure 7</xref>). Subsequently, PCA was applied to gain insight into the 10 most important features of the dataset utilised in this study, including d<sub>50</sub> (mm), Duration (s), h<sub>t</sub> (m), R<sub>c</sub> (m), T<sub>m,deep</sub> (s), L<sub>p</sub> (m), L<sub>m</sub> (m), R<sub>c</sub>/H<sub>1/3,deep</sub>, h<sub>t</sub>/H<sub>1/3,deep</sub> and I<sub>r</sub> (terms are explained in the glossary). The data corresponding to the features proposed by FSFS are selected as predictive model input for the training and testing phase of the ML algorithms.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Variation of performance metric (CV score) with the number of features during Forward Sequential Feature Selection (FSFS).</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g007.tif"/>
</fig>
<p>Further analysis of the training phase was conducted by examining the variation of RMSE in the training set (<xref ref-type="fig" rid="F8">Figure 8</xref>). The CV value and the number of training and validation iterations were set at 5 and 100, respectively. <xref ref-type="fig" rid="F8">Figure 8</xref> illustrates that, despite observing RMSE variations across all the algorithms, the average RMSE remained consistent in all the cases. This indicates that the selected algorithms in this study are capable of producing similar performance on the given dataset.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Variation of RMSE during the training phase for SVMR, ANN, RF and GBDT algorithms.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g008.tif"/>
</fig>
</sec>
</sec>
<sec id="s3-6">
<title>3.6 Evaluation metrics</title>
<p>To evaluate the performance of the machine learning algorithms in predicting relative scour depth, the predicted values were compared to the observed values using statistical metrics including the coefficient of determination (<italic>R</italic>
<sup>2</sup>), root mean square error (RMSE), mean absolute error (MAE), and relative absolute error (RAE). The Coefficient of Determination (Eq. <xref ref-type="disp-formula" rid="e5">5</xref>) describes the percentage of the dependent variable&#x2019;s fluctuation that can be predicted from the independent variables and, as such, serves as a gauge to evaluate the overall effectiveness of ML models (<xref ref-type="bibr" rid="B3">Cheng et al., 2014</xref>):<disp-formula id="e5">
<mml:math id="m20">
<mml:mrow>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>where <inline-formula id="inf16">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are the observed values, predicted values, and mean of all observed values, respectively.</p>
<p>The standard deviations between the observed and predicted values are reflected in the Root Mean Square Error (<italic>RMSE</italic>) calculated from Eq. <xref ref-type="disp-formula" rid="e6">6</xref>, and discrepancies between these values, averaged across the number of observations, is expressed in terms of the Mean Absolute Error (MAE) as in Eq. <xref ref-type="disp-formula" rid="e7">7</xref>:<disp-formula id="e6">
<mml:math id="m22">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
<disp-formula id="e7">
<mml:math id="m23">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mi>N</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mi>A</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>q</mml:mi>
<mml:mi>P</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>where, q<sub>A</sub> and q<sub>P</sub> are the actual and predicted relative scour depths, respectively.</p>
<p>In a regression test, the null hypothesis is that all of the regression coefficients are zero, i.e., the model is not predictive. The <italic>F-test</italic> is performed to determine whether accept or reject the null hypothesis. The <italic>F-test</italic> assesses whether the addition of predictor or dependent variables improves the model compared to a model with only an intercept (zero predictor variables). It quantifies the ratio of explained variance to unexplained variance (residuals) as (Eq. <xref ref-type="disp-formula" rid="e8">8</xref>):<disp-formula id="e8">
<mml:math id="m24">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:mfrac>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where, <inline-formula id="inf17">
<mml:math id="m25">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>R</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> , <inline-formula id="inf18">
<mml:math id="m26">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> , <italic>k</italic> and n are the numbers of independent variables and observations, respectively.</p>
<p>The plot of the residuals or the Discrepancy Ratio (DR) against the predicted values is also an important criteria about the relevancy of a prediction model and the residuals should ideally exhibit zero correlation with the predicted values (<xref ref-type="bibr" rid="B42">Sahay and Dutta, 2009</xref>; <xref ref-type="bibr" rid="B47">Salauddin et al., 2023</xref>).</p>
<p>The study of <xref ref-type="bibr" rid="B24">Kissell and Poserina, (2017)</xref> suggested that the statistical significance of regression models (where predicted values are compared against observed ones) should be holistically evaluated in terms of the <italic>r</italic>
<sup>
<italic>2</italic>
</sup> score, <italic>F-test</italic> score and the <italic>p</italic>-value. The values obtained from these statistical parameters should be in agreement to deduce the stability and accuracy of regression models.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s4">
<title>4 Results and discussion</title>
<sec id="s4-1">
<title>4.1 Model performances</title>
<p>The experimental dataset of <xref ref-type="bibr" rid="B45">Salauddin and Pearson (2019a)</xref> was deployed for training and testing of all the ML algorithms examined in this study following scalar transformations and feature selection. Training and testing of the models followed a common methodology which provided the basis for comparing modelled and measured dimensionless scour depths (S<sub>t</sub>/H<sub>1/3 deep</sub> [-]) in <xref ref-type="fig" rid="F9">Figure 9</xref>. Results indicate that all the four ML-based models tested in this study are capable of providing realistic approximation of scour depths. In-depth statistical evaluation of the predictive models is presented in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Comparison of predicted <italic>versus</italic> actual relative scour depths (&#x3d;S<sub>t</sub>/H<sub>1/3 deep</sub> [-]) for <bold>(A)</bold> SVMR<bold>, (B)</bold> ANN, <bold>(C)</bold> RF, and <bold>(D)</bold> GBDT.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g009.tif"/>
</fig>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Prediction evaluation metrics and statistical scores for ML-based models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Algorithm</th>
<th align="center">Coefficient of determination (<italic>r</italic>
<sup>
<italic>2</italic>
</sup>)</th>
<th align="center">Root mean square error (RMSE)</th>
<th align="center">
<italic>F-test</italic> score [<italic>F-critical</italic> &#x3d; 4.15; at <italic>p-value</italic>: 0.05]</th>
<th align="center">Mean absolute error (MAE)</th>
<th align="center">Pearson (<italic>R</italic>)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">RF</td>
<td align="center">0.62</td>
<td align="center">0.33</td>
<td align="center">82.50</td>
<td align="center">0.22</td>
<td align="center">0.95</td>
</tr>
<tr>
<td align="center">GBDT</td>
<td align="center">0.62</td>
<td align="center">0.32</td>
<td align="center">76.89</td>
<td align="center">0.19</td>
<td align="center">0.92</td>
</tr>
<tr>
<td align="center">SVMR</td>
<td align="center">0.74</td>
<td align="center">0.28</td>
<td align="center">84.20</td>
<td align="center">0.17</td>
<td align="center">0.96</td>
</tr>
<tr>
<td align="center">ANN</td>
<td align="center">0.68</td>
<td align="center">0.30</td>
<td align="center">71.50</td>
<td align="center">0.17</td>
<td align="center">0.96</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Notably, a number of data points for smaller (near to 0.0) relative scour depths fall outside the 95% CIs. This pattern is also evident for a few datapoints of large relative scour depth, while a few data points representing larger relative scour depths were inside the 95% CI zone. This suggests that while the algorithms were capable of robust overall predictions, but in the case of both smaller and larger relative scour depths, they exhibited some inconsistency. However, predictions for larger relative scour depths were more accurate (positioning on or very close to the regression line in <xref ref-type="fig" rid="F9">Figure 9</xref>). The scatter in the graphs can be explained by the Pearson R score. Among the tested algorithms, ANN, GBDT, and RF showed similar scatter with relatively lower R scores compared to SVMR. SVMR, in particular, demonstrated comparatively more accurate predictions, as reflected by the highest <italic>r</italic>
<sup>2</sup> and R scores of 0.74 and 0.85, respectively. The RMSE values of the algorithms did not vary by a large margin with respect to one another. The SVMR yielded the lowest RMSE value of 0.28 while that of the RF was the highest at 0.33. The highest RMSE value was approximately 22% of the predicted maximum relative scour depth. From a computational efficiency perspective, under the given hyperparameter conditions (<xref ref-type="table" rid="T1">Table 1</xref>) and using a computer with an 8 cores CPU, 16&#xa0;GB RAM, and 6&#xa0;GB of dedicated GPU memory, the SVMR, ANN, GBDT and RF algorithms completed the prediction task (for the test set) in 2.5, 6.93, 14.83 and 22.3&#xa0;s, respectively. This information suggests that SVMR outperforms the other algorithms in terms of computational efficiency.</p>
<p>Comprehensive statistical analyses of the developed ML-based models&#x2019; performance were conducted in this study. Statistical scores are then used to rank the performance of the four tested ML algorithms in predicting scour depth at sloping structures with shingle foreshore. <xref ref-type="table" rid="T2">Table 2</xref> shows the results of performance evaluation of the algorithms according to the criteria outlined in <xref ref-type="sec" rid="s4-1">Section 4.1</xref>.</p>
<p>The results from the evaluation metrics indicate that all the algorithms yielded strong <italic>r</italic>
<sup>
<italic>2</italic>
</sup> scores (<italic>r</italic>
<sup>
<italic>2</italic>
</sup> scores &#x3e;0.40; <xref ref-type="bibr" rid="B24">Kissell and Poserina, 2017</xref>). Hence, the <italic>F-test</italic> was performed and it was observed that all the models yielded F-score higher than the critical <italic>F-test</italic> score of 4.15 (<xref ref-type="table" rid="T2">Table 2</xref>) and also the <italic>p-values</italic> for all the models were substantially (&#x223c;10<sup>&#x2212;6</sup>) lower than the significance level of 0.05. These findings reflect the statistical significance of the results obtained from the ML algorithms and it can be inferred that the variations in the independent variable (actual relative scour depth) were accounted for by the dependent variable (predicted relative scour depth). The cumulative number of outliers in the models is expressed in the form of the RMSE. The SVMR algorithm yielded the predicted relative scour depth quantity with the smallest number of outliers, reflected in the lowest RMSE of 0.28 across all the tested algorithms. The RMSE of the other algorithms is not shown to differ significantly, suggesting the appropriateness and robustness of the proposed ML algorithms for predicting relative scour depth. A higher RMSE and MAE was coupled with lower <italic>r</italic>
<sup>
<italic>2</italic>
</sup> and <italic>vice versa</italic> for all the models. It is noted that the scale of MAE is dependent on the scale of the outputs (here, the predicted relative scour depth). The maximum and minimum absolute relative scour depth in the test set was 1.5 and 0.8, respectively, giving a mean relative scour depth of 1.15. The maximum MAE of 0.22 across the models was observed for the RF model. Conversely, the minimum MAE of 0.17 was determined for the SVMR model. Therefore, the range of MAE evaluated for this study was between 14.7% and 19% for the mean relative scour depth in the test set, consisted of 32 observations derived from the original set of 120 observations using a train-test split of 70%&#x2013;30%. The significance of the MAE analysis is that the models were able to predict the relative scour depth with an approximate accuracy of 80%. Overall, the most accurate scour predictions were attributed to the SVMR model, with the least accurate predictions being associated with the RF and GBDT models, suggesting that DT based algorithms may be less suited for obtaining predictions from smaller datasets.</p>
</sec>
<sec id="s4-2">
<title>4.2 Feature importance</title>
<p>The method of evaluating the relative contribution of various features, also referred to as variables or predictors, in a predictive model is known as Feature Importance (FI). It is useful for selecting features, comprehending the underlying data, and getting new perspectives on the subject at hand. FI reveals which features have the most impact on the model&#x2019;s predictions, essentially bridging the findings from ML to the physical consistency of the underlying processes (i.e., scouring in this study). The FI results are reported in two formats here, namely, the magnitude of the coefficients method and the permutation importance method. This is due to the fact that although the DT-based algorithms (i.e., RF and GBDT) had in-built FI analysis functions, the other two algorithms (SVMR and ANN) did not possess this function in <italic>Scikit-Learn&#x2019;s</italic> module. In the magnitude of the coefficients method, the size of the coefficients directly reflects the significance of the feature. Greater absolute values imply greater significance of the predictors. The permutation importance method involves permuting a predictor&#x2019;s values at random and analyzing the impact on model performance. The more performance is lost, the more significant the feature is thought to be. The results are reported in a similar format to that of the magnitude of coefficients method. <xref ref-type="fig" rid="F10">Figure 10</xref> summarizes the impact of the predictors on the prediction analysis.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Feature Importance Analysis showing the impact of predictors.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g010.tif"/>
</fig>
<p>In some experiments related to the measurement of relative scour depth at sloping walls with gravel foreshore, it was reported that the Iribarren Number <italic>I</italic>
<sub>
<italic>r</italic>
</sub> had a strong positive correlation with the measured scour depths for a given relative toe water depth (<italic>h</italic>
<sub>
<italic>t</italic>
</sub>
<italic>/L</italic>
<sub>
<italic>0m</italic>
</sub>) (<xref ref-type="bibr" rid="B46">Salauddin and Pearson, 2019b</xref>). Hence, it was expected that <italic>I</italic>
<sub>
<italic>r</italic>
</sub> would have the maximum influence in the prediction analysis to ensure consistency with experimental results. The FI analysis results show that 3 out of 4 (i.e., ANN, SVMR, and RF) algorithms identified <italic>I</italic>
<sub>
<italic>r</italic>
</sub> as the most important predictor. For the GBDT algorithm, <italic>I</italic>
<sub>
<italic>r</italic>
</sub> is ranked as one of the top three predictors, while the water depth at the toe of the structure (<italic>h</italic>
<sub>
<italic>t</italic>
</sub>) is identified as the most important predictor. In <xref ref-type="fig" rid="F10">Figure 10</xref>, the bars labelled as &#x2018;others&#x2019;, comprises the summation of the magnitude of importance of features including <italic>d</italic>
<sub>
<italic>50</italic>
</sub>
<italic>, Duration, R</italic>
<sub>
<italic>c</italic>
</sub>
<italic>, T</italic>
<sub>
<italic>m</italic>
</sub> <sub>
<italic>deep</italic>
</sub>
<italic>, and L</italic>
<sub>
<italic>m</italic>
</sub> from the four tested algorithms. Therefore, it can be inferred from the results of FI, that the physical scouring processes are reasonably well-captured in the proposed ML-based models.</p>
</sec>
<sec id="s4-3">
<title>4.3 Residuals</title>
<p>The residual plot for all of the tested algorithms is shown in <xref ref-type="fig" rid="F11">Figure 11</xref>
<bold>.</bold> The residuals are independent of the predicted values, highlighting that the results are in good agreement regarding the reliability of the models.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Variation of Residuals with predicted relative scour depth.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g011.tif"/>
</fig>
</sec>
<sec id="s4-4">
<title>4.4 Taylor&#x2019;s diagram</title>
<p>An effective visual method to describe the statistical metrics from predictive models is the Taylor&#x2019;s Diagram (<xref ref-type="bibr" rid="B52">Taylor, 2001</xref>). The Taylor&#x2019;s Diagram (<xref ref-type="fig" rid="F12">Figure 12</xref>) shows three statistical parameters, including the correlation coefficient projected as an azimuthal angle (in black), the radially plotted Centered Root Mean Square (cRMS) (in green), and the horizontally plotted standard deviations (in blue). Taylor diagram is particularly robust for assessing and comparing several performance aspectsof complicated models.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>Taylor&#x2019;s Diagram of the statistical metrics determined for all tested ML-based models.</p>
</caption>
<graphic xlink:href="fbuil-10-1343398-g012.tif"/>
</fig>
<p>Due to the fact that they are both the square roots of squared differences between the actual and predicted values, standard deviation and the cRMS are comparable. However, they differ from one another in the context that RMSE is used to gauge the gap between actual values and the corresponding predictions while standard deviation accounts for the spread of data around the mean. The error of prediction, or the quantitative deviation is measured using the cRMS. Here, the cRMS of the SVMR is the lowest, while the standard deviation is the highest. This essentially means that while the predicted results are more spread across the regression line, the quantity of spread is small, indicated by the low cRMS score. Conversely, RF and ANN has lower standard deviation, but the quantity of deviation is high which is reflected by the higher cRMS score. The SVMR model also yielded the highest correlation coefficient of 0.96 followed by that of ANN (0.955), RF (0.95) and GBDT (0.92). Therefore, from a holistic point of view it could be inferred from the Taylor&#x2019;s Diagram that the SVMR produced the more accurate values of predicted relative scour depth.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>Climate change-induced extreme climatic events intensify scouring in front of coastal infrastructures, posing a significant threat to their structural integrity and reliability. The development of robust prediction tools for coastal scouring is crucial for enhancing coastal resilience and safeguard these vital defences. This study examined the capabilities of advanced ML techniques for prediction of relative scour depths at sloping seawalls with shingle foreshores. This study developed a methodological framework for implementing of ML-based models for accurate predictions of relative scour depths at sloping walls with shingle foreshore. Four ML algorithms including RF, GBDT, SVMR, and ANN were utilised and tested on an experimental dataset of scour depths. We proposed a robust and efficient framework including detailed procedures for data scaling, feature selection, and tuning of the modelling parameters.</p>
<p>A methodological approach is proposed for pre-processing the physical modelling dataset to conduct missing value imputations, feature transformation (PCA), selection, and data scaling to ensure redundant data and missing values do not impair the performance of the ML models. In order to verify the ML algorithms on a randomly selected sub-set of training data, cross validation was carried out in the training step. A typical train-test split of 70%&#x2013;30% was implemented. These precautions ensured that a consistent methodology was followed to achieve comparable outcomes from the predictions made by the four algorithms. Iribarren Number (Ir) was identified as the most important parameter influencing the scouring process, in agreement with the physical process of scouring.</p>
<p>The performance of the proposed ML-based predictive models were evaluated for a comprehensive experimental dataset. The predicted relative scour depth and comprehensive statistical evaluation confirmed the robust performance and accuracy of all the tested algorithms.</p>
<p>A set of statistical indices, (<italic>r</italic>
<sup>
<italic>2</italic>
</sup>, RMSE, MAE, <italic>F-test</italic> and Pearson <italic>R</italic>) were incorporated to gauge the efficiency of the tested ML algorithms. The SVMR algorithm showed superior performance compared to the other tested algorithms with an <italic>r</italic>
<sup>
<italic>2</italic>
</sup> score of 0.74, RMSE of 0.28, MAE of 0.17 and Pearson R value of 0.96. The DT based algorithms were not able to match performance of SVMR and ANN with scores of 0.62 for <italic>r</italic>
<sup>
<italic>2</italic>
</sup> for both RF and GBDT. ANN was identified as the second-best performing algorithm with a <italic>r</italic>
<sup>
<italic>2</italic>
</sup> score closest to that of SVMR the (0.68). The <italic>F-test</italic> score and the Pearson <italic>R</italic> values of the algorithms are indicative of the fact that the variation of the independent variable is accounted for by the dependent variables or the predictors and that there is strong correlation between the actual and predicted values. These findings were reinforced by the high Pearson <italic>R</italic> values of 0.96, 0.955, 0.95 and 0.92 for SVMR, ANN, RF and GBDT, respectively. The SVMR model was also the most computationally efficient model (&#x3c;3s), more than two times faster than ANN (6.93s) followed by DT based GBDT (14.83s) and RF (22.3s). The comparison of MAE revealed that accuracy of predictions was over 80% for all the algorithms. One important reason of DTs underperforming in this study may be due to the relatively small number of training data. Although there is no explicit requirement of the amount of training data required by ML algorithms, larger and more diverse datasets could improve the performance of ML-based models presented in this study. Future studies should focus on further improving the performance of the proposed predictive tool by the inclusion of larger experimental datasets. Hybrid machine learning approaches with optimisation techniques could potentially enhance predictive performance of the models proposed here and should be tested on wave-induced scouring datasets. The method proposed in this study could be adopted by coastal engineers for rapid scour depth prediction and inform design and maintenance of coastal defence structures.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: Data will be made available on reasonable request. Requests to access these datasets should be directed to <email>md.salauddin@ucd.ie</email>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>MAH: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Software, Validation, Visualization, Writing&#x2013;original draft. SA: Writing&#x2013;review and editing. JO&#x2019;S: Writing&#x2013;review and editing, Supervision. MS: Conceptualization, Funding acquisition, Methodology, Project administration, Resources, Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Babaee</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maroufpoor</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jalali</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zarei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Elbeltagi</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Artificial intelligence approach to estimating rice yield</article-title>. <source>Irrigation Drainage</source> <volume>70</volume> (<issue>4</issue>), <fpage>732</fpage>&#x2013;<lpage>742</lpage>. <pub-id pub-id-type="doi">10.1002/ird.2566</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Feature selection in machine learning: a new perspective</article-title>. <source>Neurocomputing</source> <volume>300</volume>, <fpage>70</fpage>&#x2013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2017.11.077</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname>
<given-names>C.-L.</given-names>
</name>
<name>
<surname>Shalabh</surname>
</name>
<name>
<surname>Garg</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Coefficient of determination for multiple measurement error models</article-title>. <source>J. Multivar. Analysis</source> <volume>126</volume>, <fpage>137</fpage>&#x2013;<lpage>152</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmva.2014.01.006</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>den Bieman</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>van Gent</surname>
<given-names>M. R. A.</given-names>
</name>
<name>
<surname>van den Boogaard</surname>
<given-names>H. F. P.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>Wave overtopping predictions using an advanced machine learning technique</article-title>. <source>Coast. Eng.</source> <volume>166</volume>, <fpage>103830</fpage>. <pub-id pub-id-type="doi">10.1016/j.coastaleng.2020.103830</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>den Bieman</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Wilms</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>van den Boogaard</surname>
<given-names>H. F. P.</given-names>
</name>
<name>
<surname>van Gent</surname>
<given-names>M. R. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Prediction of mean wave overtopping discharge using gradient boosting decision trees</article-title>. <source>Water</source> <volume>12</volume> (<issue>6</issue>), <fpage>1703</fpage>. <pub-id pub-id-type="doi">10.3390/w12061703</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Donnelly</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Daneshkhah</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Forecasting global climate drivers using Gaussian processes and convolutional autoencoders</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>128</volume>, <fpage>107536</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2023.107536</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elbeltagi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kumari</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Dharpure</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mokhtar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Alsafadi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Prediction of combined terrestrial evapotranspiration index (CTEI) over large river basin based on machine learning approaches</article-title>. <source>Water</source> <volume>13</volume> (<issue>4</issue>), <fpage>547</fpage>. <pub-id pub-id-type="doi">10.3390/w13040547</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elbeltagi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pande</surname>
<given-names>C. B.</given-names>
</name>
<name>
<surname>Kouadri</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Islam</surname>
<given-names>A. R. M. T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Applications of various data-driven models for the prediction of groundwater quality index in the Akot basin, Maharashtra, India</article-title>. <source>Environ. Sci. Pollut. Res.</source> <volume>29</volume> (<issue>12</issue>), <fpage>17591</fpage>&#x2013;<lpage>17605</lpage>. <pub-id pub-id-type="doi">10.1007/s11356-021-17064-7</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elbisy</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Machine learning techniques for estimating wave-overtopping discharges at coastal structures</article-title>. <source>Ocean. Eng.</source> <volume>273</volume>, <fpage>113972</fpage>. <pub-id pub-id-type="doi">10.1016/j.oceaneng.2023.113972</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Elbisy</surname>
<given-names>M. S.</given-names>
</name>
<name>
<surname>Elbisy</surname>
<given-names>A. M. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Prediction of significant wave height by artificial neural networks and multiple additive regression trees</article-title>. <source>Ocean. Eng.</source> <volume>230</volume>, <fpage>109077</fpage>. <pub-id pub-id-type="doi">10.1016/j.oceaneng.2021.109077</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>EurOtop</surname>
</name>
</person-group> (<year>2018</year>). <source>Manual on Wave Overtopping of Sea Defences and Related Structures</source>. <edition>2nd Edn</edition>. <comment>Available online at: <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="http://www.overtopping-manual.com">www.overtopping-manual.com</ext-link>
</comment> (<comment>accessed July, 2023</comment>)</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fitri</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hashim</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Abdul Maulud</surname>
<given-names>K. N.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Dynamics of sediment transport and erosion-deposition patterns in the locality of a detached low-crested breakwater on a cohesive coast</article-title>. <source>Water</source> <volume>11</volume> (<issue>8</issue>), <fpage>1721</fpage>. <pub-id pub-id-type="doi">10.3390/w11081721</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Formentin</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Zanuttigh</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>van der Meer</surname>
<given-names>J. W.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A neural network tool for predicting wave reflection, overtopping and transmission</article-title>. <source>Coast. Eng. J.</source> <volume>59</volume> (<issue>1</issue>), <fpage>1750006</fpage>. <pub-id pub-id-type="doi">10.1142/S0578563417500061</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Fowler</surname>
<given-names>J. E.</given-names>
</name>
</person-group> (<year>1992</year>). &#x201c;<article-title>Scour problems and methods for prediction of maximum scour at vertical seawalls</article-title>,&#x201d; in <source>Us army corps of engineers</source> (<publisher-loc>Vicksburg, MS, USA</publisher-loc>: <publisher-name>Coastal Engineering Research Center</publisher-name>). <comment>W. E. S. (eds.), Technical Report CERC-92&#x2013;16</comment>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ghiasi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Noori</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sheikhian</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zeynolabedin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jun</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Uncertainty quantification of granular computing-neural network model for prediction of pollutant longitudinal dispersion coefficient in aquatic streams</article-title>. <source>Sci. Rep.</source> <volume>12</volume>, <fpage>4610</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-08417-4</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Glorot</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Understanding the difficulty of training deep feedforward neural networks</article-title>,&#x201d; in <conf-name>International conference on artificial intelligence and statistics. 2010</conf-name>.</citation>
</ref>
<ref id="B16">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Habib</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>O&#x27;Sullivan</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023b</year>). &#x201c;<article-title>Prediction of wave overtopping rates at sloping structures using artificial intelligence</article-title>,&#x201d; in <conf-name>Proceedings of the 40th IAHR World Congress. Rivers&#x2013;Connecting Mountains and Coasts</conf-name>, <fpage>404</fpage>&#x2013;<lpage>413</lpage>. <pub-id pub-id-type="doi">10.3850/978-90-833476-1-5_iahr40wc-p0115-cd</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Habib</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>O&#x27;Sullivan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022a</year>). <source>Comparison of machine learning algorithms in predicting wave overtopping discharges at vertical breakwaters</source>. <publisher-loc>Austria</publisher-loc>: <publisher-name>EGU General Assembly Vienna</publisher-name>, <fpage>EGU22</fpage>&#x2013;<lpage>329</lpage>. <comment>23&#x2013;27 May 2022</comment>. <pub-id pub-id-type="doi">10.5194/egusphere-egu22-329</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Habib</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>O&#x2019;Sullivan</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>Enhanced wave overtopping simulation at vertical breakwaters using machine learning algorithms</article-title>. <source>PLOS ONE</source> <volume>18</volume> (<issue>8</issue>), <fpage>e0289318</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0289318</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Habib</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>O&#x2019;Sullivan</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Prediction of wave overtopping characteristics at coastal flood defences using machine learning algorithms: a systematic rreview</article-title>. <source>IOP Conf. Ser. Earth Environ. Sci.</source> <volume>1072</volume> (<issue>1</issue>), <fpage>012003</fpage>. <pub-id pub-id-type="doi">10.1088/1755-1315/1072/1/012003</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>G. B.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Extreme learning machine for regression and multiclass classification</article-title>. <source>IEEE Trans. Syst. Man. Cyber Part B</source> <volume>42</volume>, <fpage>513</fpage>&#x2013;<lpage>529</lpage>. <pub-id pub-id-type="doi">10.1109/tsmcb.2011.2168604</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kawashima</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Kumano</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Prediction of mind-wandering with electroencephalogram and non-linear regression modeling</article-title>. <source>Front. Hum. Neurosci.</source> <volume>11</volume>, <fpage>365</fpage>. <pub-id pub-id-type="doi">10.3389/fnhum.2017.00365</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khosravi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rezaie</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Cooper</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Kalantari</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hatamiafkoueieh</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Soil water erosion susceptibility assessment using deep learning algorithms</article-title>. <source>J. Hydrology</source> <volume>618</volume>, <fpage>129229</fpage>. <pub-id pub-id-type="doi">10.1016/j.jhydrol.2023.129229</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kissell</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Poserina</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Regression models</article-title>,&#x201d; in <source>Optimal sports math, statistics, and fantasy</source> (<publisher-name>Elsevier</publisher-name>), <fpage>39</fpage>&#x2013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-12-805163-4.00002-5</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kotu</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Deshpande</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Classification</article-title>,&#x201d; in <source>Predictive analytics and data mining</source> (<publisher-name>Elsevier</publisher-name>), <fpage>63</fpage>&#x2013;<lpage>163</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-12-801460-8.00004-5</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Parameter prediction of the non-linear nomoto model for different ship loading conditions using support vector regression</article-title>. <source>J. Mar. Sci. Eng.</source> <volume>11</volume> (<issue>5</issue>), <fpage>903</fpage>. <pub-id pub-id-type="doi">10.3390/jmse11050903</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source> <volume>521</volume>, <fpage>436</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Motoda</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2012</year>). <source>Feature selection for knowledge discovery and data mining</source>. <publisher-name>Springer Science &#x26; Business Media</publisher-name>.</citation>
</ref>
<ref id="B30">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Marcano-Cedeno</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Quintanilla-Dominguez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cortina-Januchs</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Andina</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Feature selection using sequential forward selection and classification applying artificial metaplasticity neural network</article-title>,&#x201d; in <conf-name>IECON 2010 - 36th Annual Conference on IEEE Industrial Electronics Society</conf-name>, <fpage>2845</fpage>&#x2013;<lpage>2850</lpage>. <pub-id pub-id-type="doi">10.1109/IECON.2010.5675075</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>M&#xfc;ller</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Allsop</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Bruce</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kortenhaus</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pearce</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sutherland</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). &#x201c;<article-title>The occurrence and effects of wave impacts</article-title>,&#x201d; in <conf-name>Proceedings of the ICE-Maritime Engineering (ICE)</conf-name>, <fpage>167</fpage>&#x2013;<lpage>173</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Noori</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ghiasi</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Salehi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Esmaeili Bidhendi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Raeisi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Partani</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>An efficient data driven-based model for prediction of the total sediment load in rivers</article-title>. <source>Hydrology</source> <volume>9</volume> (<issue>2</issue>), <fpage>36</fpage>. <pub-id pub-id-type="doi">10.3390/hydrology9020036</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedregosa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Varoquaux</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Gramfort</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michel</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Thirion</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Grisel</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Scikit-learn: machine learning in Python</article-title>. <pub-id pub-id-type="doi">10.48550/arXiv.1201.0490</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q. P.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A partial cell technique for modeling the morphological change and scour</article-title>. <source>Coast. Eng.</source> <volume>131</volume>, <fpage>88</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1016/j.coastaleng.2017.09.006</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Q. P.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Impulsive wave overtopping with toe scour at a vertical seawall</article-title>,&#x201d; in <conf-name>ICE Breakwater Conference 2023</conf-name>, <conf-loc>UK</conf-loc>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pourzangbar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brocchini</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Saber</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mahjoobi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mirzaaghasi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Barzegar</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017b</year>). <article-title>Prediction of scour depth at breakwaters due to non-breaking waves using machine learning approaches</article-title>. <source>Appl. Ocean. Res.</source> <volume>63</volume>, <fpage>120</fpage>&#x2013;<lpage>128</lpage>. <pub-id pub-id-type="doi">10.1016/j.apor.2017.01.012</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pourzangbar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Losada</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Saber</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ahari</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>Larroud&#xe9;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Vaezi</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2017a</year>). <article-title>Prediction of non-breaking wave induced scour depth at the trunk section of breakwaters using Genetic Programming and Artificial Neural Networks</article-title>. <source>Coast Eng.</source> <volume>121</volume>, <fpage>107</fpage>&#x2013;<lpage>118</lpage>. <pub-id pub-id-type="doi">10.1016/j.coastaleng.2016.12.008</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Powell</surname>
<given-names>K. A.</given-names>
</name>
<name>
<surname>Lowe</surname>
<given-names>J. P.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>The scouring of sediments at the toe of seawalls</article-title>. <source>In: Proceedings of the Hornafjordor International Coastal Symposium, Iceland</source>, <fpage>749</fpage>&#x2013;<lpage>755</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raikar</surname>
<given-names>R. V.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.-Y.</given-names>
</name>
<name>
<surname>Shih</surname>
<given-names>H.-P.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>J.-H.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Prediction of contraction scour using ANN and GA</article-title>. <source>Flow Meas. Instrum.</source> <volume>50</volume>, <fpage>26</fpage>&#x2013;<lpage>34</lpage>. <pub-id pub-id-type="doi">10.1016/j.flowmeasinst.2016.06.006</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Roessner</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Nahid</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chapman</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hunter</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bellgard</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Metabolomics &#x2013; the combination of analytical biochemistry, biology, and informatics</article-title>,&#x201d; in <source>Comprehensive biotechnology</source> (<publisher-name>Elsevier</publisher-name>), <fpage>435</fpage>&#x2013;<lpage>447</lpage>. <pub-id pub-id-type="doi">10.1016/B978-0-444-64046-8.00027-6</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roushangar</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Koosheh</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Evaluation of GA-SVR method for modeling bed load transport in gravel-bed rivers</article-title>. <source>J. Hydrology</source> <volume>527</volume>, <fpage>1142</fpage>&#x2013;<lpage>1152</lpage>. <pub-id pub-id-type="doi">10.1016/j.jhydrol.2015.06.006</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sahay</surname>
<given-names>R. R.</given-names>
</name>
<name>
<surname>Dutta</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Prediction of longitudinal dispersion coefficients in natural rivers using genetic algorithm</article-title>. <source>Hydrology Res.</source> <volume>40</volume> (<issue>6</issue>), <fpage>544</fpage>&#x2013;<lpage>552</lpage>. <pub-id pub-id-type="doi">10.2166/nh.2009.014</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>O&#x2019;Sullivan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pearson</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>New insights in the probability distributions of wave-by-wave overtopping volumes at vertical breakwaters</article-title>. <source>Sci. Rep.</source> <volume>12</volume>, <fpage>16228</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-20464-5</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pearson</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A laboratory study on wave overtopping at vertical seawalls with a shingle foreshore</article-title>. <source>Coast. Eng. Proc.</source> (<issue>36</issue>), <fpage>56</fpage>. <pub-id pub-id-type="doi">10.9753/icce.v36.waves.56</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pearson</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2019a</year>). <article-title>Wave overtopping and toe scouring at a plain vertical seawall with shingle foreshore: a Physical model study</article-title>. <source>Ocean. Eng.</source> <volume>171</volume>, <fpage>286</fpage>&#x2013;<lpage>299</lpage>. <pub-id pub-id-type="doi">10.1016/j.oceaneng.2018.11.011</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pearson</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2019b</year>). <article-title>Experimental study on toe scouring at sloping walls with gravel foreshores</article-title>. <source>J. Mar. Sci. Eng.</source> <volume>7</volume>, <fpage>198</fpage>. <pub-id pub-id-type="doi">10.3390/jmse7070198</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shaffrey</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Habib</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Data-driven approaches in predicting scour depths at a vertical seawall on a permeable shingle foreshore</article-title>. <source>J. Coast Conserv.</source> <volume>27</volume>, <fpage>18</fpage>. <pub-id pub-id-type="doi">10.1007/s11852-023-00948-w</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salauddin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pearson</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Laboratory investigation of overtopping at a sloping structure with permeable shingle foreshore</article-title>. <source>Ocean Engineering</source> <volume>197</volume>. <pub-id pub-id-type="doi">10.1016/j.oceaneng.2019.106866</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sutherland</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Brampton</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Motyka</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Blanco</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Whitehouse</surname>
<given-names>R. J. W.</given-names>
</name>
</person-group> (<year>2003</year>). <source>Beach lowering in front of coastal structures-Research Scoping Study</source>. <publisher-loc>London, UK</publisher-loc>. <comment>Report FD1916/TR</comment>.</citation>
</ref>
<ref id="B49">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sutherland</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Brampton</surname>
<given-names>A. H.</given-names>
</name>
<name>
<surname>Obrai</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dunn</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Whitehouse</surname>
<given-names>R. J. W.</given-names>
</name>
</person-group> (<year>2008</year>). <source>Understanding the lowering of beaches in front of coastal defence structures, Stage 2-Research Scoping Study</source>. <publisher-loc>London, UK</publisher-loc>. <comment>Report FD1927/TR</comment>.</citation>
</ref>
<ref id="B50">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sutherland</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Obhrai</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Whitehouse</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pearce</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). &#x201c;<article-title>Laboratory tests of scour at a seawall</article-title>,&#x201d; in <conf-name>Proceedings of the 3rd International Conference on Scour and Erosion, CURNET</conf-name> (<publisher-loc>Gouda, Netherlands</publisher-loc>: <publisher-name>Technical University of Denmark</publisher-name>).</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sutton</surname>
<given-names>C. D.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Classification and regression trees, bagging, and boosting</article-title> (pp. <fpage>303</fpage>&#x2013;<lpage>329</lpage>). <pub-id pub-id-type="doi">10.1016/S0169-7161(04)24011-1</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taylor</surname>
<given-names>K. E.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Summarizing multiple aspects of model performance in a single diagram</article-title>. <source>J. Geophys. Res. Atmos.</source> <volume>106</volume> (<issue>D7</issue>), <fpage>7183</fpage>&#x2013;<lpage>7192</lpage>. <pub-id pub-id-type="doi">10.1029/2000JD900719</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tseng</surname>
<given-names>I.-F.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>C.-H.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>P.-H.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>T.-C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Physical mechanism for seabed scouring around a breakwater&#x2014;a case study in mailiao port</article-title>. <source>J. Mar. Sci. Eng.</source> <volume>10</volume> (<issue>10</issue>), <fpage>1386</fpage>. <pub-id pub-id-type="doi">10.3390/jmse10101386</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Verhaeghe</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>De Rouck</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>van der Meer</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Combined classifier&#x2013;quantifier model: a 2-phases neural model for prediction of wave overtopping at coastal structures</article-title>. <source>Coast. Eng.</source> <volume>55</volume> (<issue>5</issue>), <fpage>357</fpage>&#x2013;<lpage>374</lpage>. <pub-id pub-id-type="doi">10.1016/j.coastaleng.2007.12.002</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wallis</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Whitehouse</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Lyness</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Development of guidance for the management of the toe of coastal defence structures. In Coasts, marine structures and breakwaters: Adapting to change: Proceedings of the 9th international conference organised by the Institution of Civil Engineers and held in Edinburgh on 16 to 18 September 2009</article-title>. <source>Thomas Telford Ltd.</source>, <fpage>696</fpage>&#x2013;<lpage>707</lpage>.</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Low</surname>
<given-names>Y. M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>C.-H.</given-names>
</name>
<name>
<surname>Chiew</surname>
<given-names>Y.-M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Numerical simulation of scour around a submarine pipeline using computational fluid dynamics and discrete element method</article-title>. <source>Appl. Math. Model.</source> <volume>55</volume>, <fpage>400</fpage>&#x2013;<lpage>416</lpage>. <pub-id pub-id-type="doi">10.1016/j.apm.2017.10.007</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yeganeh-Bakhtiary</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>EyvazOghli</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shabakhty</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Machine learning prediction of wave characteristics: comparison between semi-empirical approaches and DT model</article-title>. <source>Ocean. Eng.</source> <volume>286</volume> (<issue>2</issue>), <fpage>115583</fpage>. <pub-id pub-id-type="doi">10.1016/j.oceaneng.2023.115583</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yeganeh-Bakhtiary</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>EyvazOghli</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Shabakhty</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kamranzad</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Machine learning as a downscaling approach for prediction of wind characteristics under future climate change scenarios</article-title>. <source>Complexity</source> <volume>2022</volume>, <fpage>8451812</fpage>. <pub-id pub-id-type="doi">10.1155/2022/8451812</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yeganeh-Bakhtiary</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Houshangi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Abolfathi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Lagrangian two-phase flow modeling of scour in front of vertical breakwater</article-title>. <source>Coast. Eng. J.</source> <volume>62</volume> (<issue>2</issue>), <fpage>252</fpage>&#x2013;<lpage>266</lpage>. <pub-id pub-id-type="doi">10.1080/21664250.2020.1747140</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zanuttigh</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Formentin</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>van der Meer</surname>
<given-names>J. W.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Prediction of extreme and tolerable wave overtopping discharges through an advanced neural network</article-title>. <source>Ocean. Eng.</source> <volume>127</volume>, <fpage>7</fpage>&#x2013;<lpage>22</lpage>. <pub-id pub-id-type="doi">10.1016/j.oceaneng.2016.09.032</pub-id>
</citation>
</ref>
</ref-list>
<sec id="s11">
<title>Glossary</title>
<table-wrap id="udT1" position="float">
<table>
<tbody valign="top">
<tr>
<td align="left">d<sub>50</sub>
</td>
<td align="left">Effective particle size (mm)</td>
</tr>
<tr>
<td align="left">Duration</td>
<td align="left">Storm duration (s)</td>
</tr>
<tr>
<td align="left">h<sub>t</sub>
</td>
<td align="left">Water depth at toe of the structure (s)</td>
</tr>
<tr>
<td align="left">R<sub>c</sub>
</td>
<td align="left">Crest freeboard of the structure (m)</td>
</tr>
<tr>
<td align="left">T<sub>m,deep</sub>
</td>
<td align="left">Average Wave Period in Deep Water (s)Angle of wave attack (<sup>o</sup>)</td>
</tr>
<tr>
<td align="left">L<sub>p</sub>
</td>
<td align="left">Peak Wavelength in Deep Water (m)</td>
</tr>
<tr>
<td align="left">L<sub>m</sub>
</td>
<td align="left">Mean Wavelength in Deep Water (m)</td>
</tr>
<tr>
<td align="left">R<sub>c</sub>/H<sub>1/3, deep</sub>
</td>
<td align="left">Relative Crest Freeboard (&#x2212;), where H<sub>1/3, deep</sub> is the significant wave height (m) calculated from the average of the highest one-third of all waves</td>
</tr>
<tr>
<td align="left">h<sub>t</sub>/H<sub>1/3, deep</sub>
</td>
<td align="left">Relative Water Depth (&#x2212;)</td>
</tr>
<tr>
<td align="left">S<sub>t</sub>/H<sub>1/3 deep</sub>
</td>
<td align="left">Relative Scour Depth (&#x2212;); where S<sub>t</sub> &#x3d; Scour Depth (m)</td>
</tr>
<tr>
<td align="left">I<sub>r</sub>
</td>
<td align="left">Iribaren Number (&#x2212;) [ &#x3d; <inline-formula id="inf19">
<mml:math id="m27">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="italic">tan</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:msqrt>
<mml:mfrac>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:msqrt>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula> ]; where <inline-formula id="inf20">
<mml:math id="m28">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; angle of foreshore</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</back>
</article>