<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Earth Sci.</journal-id>
<journal-title>Frontiers in Earth Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Earth Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-6463</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1223686</article-id>
<article-id pub-id-type="doi">10.3389/feart.2023.1223686</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Earth Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>CFM: a convolutional neural network for first-motion polarity classification of seismic records in volcanic and tectonic areas</article-title>
<alt-title alt-title-type="left-running-head">Messuti et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/feart.2023.1223686">10.3389/feart.2023.1223686</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Messuti</surname>
<given-names>Giovanni</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2342021/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Scarpetta</surname>
<given-names>Silvia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/10071/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Amoroso</surname>
<given-names>Ortensia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/842796/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Napolitano</surname>
<given-names>Ferdinando</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1028122/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Falanga</surname>
<given-names>Mariarosaria</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2202949/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Capuano</surname>
<given-names>Paolo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1423308/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Physics &#x201c;E.R. Caianiello&#x201d;, University of Salerno</institution>, <addr-line>Fisciano</addr-line>, <country>Italy</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Section of Naples, National Institute for Nuclear Physics (INFN)</institution>, <addr-line>Naples</addr-line>, <country>Italy</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Information and Electrical Engineering and Applied Mathematics (DIEM), University of Salerno</institution>, <addr-line>Fisciano</addr-line>, <country>Italy</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1449831/overview">Georg R&#xfc;mpker</ext-link>, Goethe University Frankfurt, Germany</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1544364/overview">Nishtha Srivastava</ext-link>, Frankfurt Institute for Advanced Studies, Germany</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2196495/overview">Darren Tan</ext-link>, University of Alaska Fairbanks, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Giovanni Messuti, <email>g.messuti@studenti.unisa.it</email>; Ortensia Amoroso, <email>oamoroso@unisa.it</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>20</day>
<month>07</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>11</volume>
<elocation-id>1223686</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>05</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>07</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Messuti, Scarpetta, Amoroso, Napolitano, Falanga and Capuano.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Messuti, Scarpetta, Amoroso, Napolitano, Falanga and Capuano</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>First-motion polarity determination is essential for deriving volcanic and tectonic earthquakes&#x2019; focal mechanisms, which provide crucial information about fault structures and stress fields. Manual procedures for polarity determination are time-consuming and prone to human error, leading to inaccurate results. Automated algorithms can overcome these limitations, but accurately identifying first-motion polarity is challenging. In this study, we present the Convolutional First Motion (CFM) neural network, a label-noise robust strategy based on a Convolutional Neural Network, to automatically identify first-motion polarities of seismic records. CFM is trained on a large dataset of more than 140,000 waveforms and achieves a high accuracy of 97.4% and 96.3% on two independent test sets. We also demonstrate CFM&#x2019;s ability to correct mislabeled waveforms in 92% of cases, even when they belong to the training set. Our findings highlight the effectiveness of deep learning approaches for first-motion polarity determination and suggest the potential for combining CFM with other deep learning techniques in volcano seismology.</p>
</abstract>
<kwd-group>
<kwd>deep convolutional neural networks</kwd>
<kwd>automatic classification</kwd>
<kwd>machine learning</kwd>
<kwd>self-organizing maps</kwd>
<kwd>volcanic and tectonic earthquakes</kwd>
<kwd>first-motion polarity</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Volcanology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>In the field of Earth sciences, the study of seismic waves generated by earthquakes occupies an important role since it allows us to retrieve the main features of both the propagation medium and the seismic source. As for the seismic source, the attention is mainly devoted to estimating the geometric and kinematic parameters, including the location, magnitude, fault dimension and focal mechanisms. Focal mechanisms are crucial to characterize the seismogenic fault structures and the stress field of a region, from local to nationwide scale, in tectonic (<xref ref-type="bibr" rid="B54">Vavry&#x10d;uk, 2014</xref>; <xref ref-type="bibr" rid="B35">Napolitano et al., 2021a</xref>; <xref ref-type="bibr" rid="B53">Uchide et al., 2022</xref>), and volcanic areas (<xref ref-type="bibr" rid="B47">Roman et al., 2006</xref>; <xref ref-type="bibr" rid="B19">Judson et al., 2018</xref>; <xref ref-type="bibr" rid="B24">La Rocca and Galluzzo, 2019</xref>; <xref ref-type="bibr" rid="B3">Aoki, 2022</xref>; <xref ref-type="bibr" rid="B57">Zhan et al., 2022</xref>).</p>
<p>The focal mechanisms can be computed using P-wave first-motion polarity (e.g., FPFIT; <xref ref-type="bibr" rid="B42">Reasenberg, 1985</xref>; <xref ref-type="bibr" rid="B50">Snoke et al., 2003</xref>; <xref ref-type="bibr" rid="B18">Hardebeck and Shearer, 2002</xref>), the waveform information (e.g., <xref ref-type="bibr" rid="B58">Zhao and Helmberger, 1994</xref>) or both (<xref ref-type="bibr" rid="B56">Weber, 2018</xref>). P-wave polarity is also used as an additional constraint in the moment-tensor inversion (e.g., in volcanic settings, <xref ref-type="bibr" rid="B8">Dahm and Brandsdottir, 1997</xref>; <xref ref-type="bibr" rid="B32">Miller et al., 1998</xref>; <xref ref-type="bibr" rid="B39">Pesicek et al., 2012</xref>; <xref ref-type="bibr" rid="B2">Alvizuri and Tape, 2016</xref>) and full waveform inversion (e.g., for explosion <xref ref-type="bibr" rid="B7">Chiang et al., 2014</xref>; <xref ref-type="bibr" rid="B13">Ford et al., 2009</xref>). Determining first-motion polarities by manual procedures, mostly done for larger events, is time-consuming, susceptible to human error and can result in different outcomes depending on the expert analyst. In addition, a proper identification of the first-motion polarity can be difficult when dealing with small magnitude earthquakes. This may be due to the unfavorable signal-to-noise ratio. An enhanced method of identifying first-motion polarities will allow us to resolve the focal mechanism of smaller magnitude events, thereby improving our ability to characterize and interpret seismogenic areas. Automated procedures (e.g., <xref ref-type="bibr" rid="B6">Chen and Holland, 2016</xref>; <xref ref-type="bibr" rid="B41">Pugh et al., 2016</xref>) can avoid drawbacks, such as time consumption and ensure reproducibility. Despite this, identifying first-motion polarity is not a straightforward classification task that can be easily expressed using mathematical procedures. Consequently, the effectiveness of the automated algorithms (not based on machine learning) relies on a limited number of parameters, which require intensive human involvement to fine-tune, and may result in worse performance compared to human analysis (<xref ref-type="bibr" rid="B48">Ross et al., 2018</xref>).</p>
<p>Deep learning offers a notable advantage in that prior knowledge of the observed phenomena is not a prerequisite for model development. This is attributed to the capability of Deep Neural Networks (DNNs) to autonomously extract significant features from raw data, eliminating the need for a mathematical representation of the problem. Moreover, when confronted with extensive datasets, deep learning has proved to be a suitable and highly effective methodology to be employed. Hence, the vast amount of seismological data represents an excellent opportunity for the application of DNNs, making deep learning an ideal choice for our purposes. Recent studies demonstrated the possibility of developing effective and competitive applications of DNNs in the study of seismic waves generated by earthquakes, volcanic eruptions, explosions, along with other sources (<xref ref-type="bibr" rid="B33">Mousavi and Beroza, 2022</xref>). DNNs have been used for events detection and location (<xref ref-type="bibr" rid="B38">Perol et al., 2018</xref>), arrival times picking (<xref ref-type="bibr" rid="B48">Ross et al., 2018</xref>; <xref ref-type="bibr" rid="B60">Zhu and Beroza, 2019</xref>), data denoising (<xref ref-type="bibr" rid="B43">Richardson and Feller, 2019</xref>), classification of volcano-seismic events (<xref ref-type="bibr" rid="B29">L&#xf3;pez-P&#xe9;rez et al., 2020</xref>), construction of suitable ontologies (<xref ref-type="bibr" rid="B12">Falanga et al., 2022</xref>), discrimination of explosive and tectonic sources (<xref ref-type="bibr" rid="B28">Linville et al., 2019</xref>; <xref ref-type="bibr" rid="B22">Kong et al., 2022</xref>), waveform recognition both focusing on transients and continuous background acquisition (<xref ref-type="bibr" rid="B44">Rincon-Yanez et al., 2022</xref>) and for ground motion prediction equations (<xref ref-type="bibr" rid="B40">Prezioso et al., 2022</xref>).</p>
<p>Several studies have demonstrated the significant applicability of Convolutional Neural Networks (<xref ref-type="bibr" rid="B25">LeCun et al., 2015</xref>) in determining the first-motion polarity. CNNs use convolutional layers to extract spatial patterns from a multi-dimensional input array or matrix-like data. By applying multiple filters with adjustable weights through a process known as convolution, these filters extract relevant features through their scanning process. Stacking multiple convolutional layers allows the network to automatically learn and identify relevant abstract features useful for the task. The ability of CNNs to capture complex spatial relationships has made them particularly effective in a wide range of image and signal processing tasks, including the determination of first-motion polarity. One of the earliest studies in this field, conducted by <xref ref-type="bibr" rid="B48">Ross et al. (2018)</xref>, involved training a simple CNN on 18.2&#xa0;million seismograms from the Southern California Seismic Network (SCSN) catalog, achieving a precision in determining polarities of 95%. <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref> established a lower limit on the number of waveforms required for a satisfactory level of performance during training. The same authors explored the possibility of using a CNN to predict waveforms deriving from events located in regions different from those where data used for the training set have been collected. <xref ref-type="bibr" rid="B52">Uchide (2020)</xref> derived focal mechanisms and important information about the stress field in Japan exploiting the first-motion polarities determined by using a CNN-based technique. <xref ref-type="bibr" rid="B27">Li et al. (2023)</xref> utilized the CNN by <xref ref-type="bibr" rid="B59">Zhao et al. (2023)</xref> to develop an automatic workflow for focal mechanism inversion.</p>
<p>In this work, we present the Convolutional First Motion (CFM) neural network, a label-noise robust strategy based on a CNN to automatically identify first-motion polarities of seismic waves. We take advantage of the regularization effects of dropout layers and the implicit regularization properties of Stochastic Gradient Descent (SGD), when used in combination with early stopping, to handle a percentage of mislabelling (often known as noisy labels). CFM is trained on more than 140,000 waveforms derived from INSTANCE dataset (<xref ref-type="bibr" rid="B31">Michelini et al., 2021</xref>), and tested both on 8,983 waveforms belonging to different events of the same dataset and on 4,072 waveforms collected from <xref ref-type="bibr" rid="B34">Napolitano et al. (2021b)</xref>. We found that when CFM is applied to mislabeled waveforms, which we identified through a data visualization procedure, it corrects them in 92% of the cases, even when they belong to the training set. CFM showed high accuracy levels (i.e., 97.4% and 96.3%) when tested on two independent test sets, high reliability and great generalization ability. The approach shown in our study reveals that an appropriate augmentation procedure can make the network able to deal with uncertainty in arrival times, which increases the potential for using CFM in combination with automatic deep learning techniques for phase picking. Such methodology is expected to have a strong impact on any problem related to the source modeling of tectonic and volcanic quakes, whose construction is founded on the best picking and phase recognition.</p>
</sec>
<sec id="s2">
<title>2 Data</title>
<p>We collected the seismic waveforms included in the INSTANCE dataset (<xref ref-type="bibr" rid="B31">Michelini et al., 2021</xref>) and used them to train the neural network and to evaluate its performance. The dataset, specifically compiled to apply machine learning techniques, comprises 1,159,249 waveforms originating from different sources (natural and anthropogenic earthquakes, volcanic eruptions, landslides along with other sources). The waveforms were registered by both velocimeters (HH, EH channels) and accelerometers (HN channel) seismometers belonging to 19 seismic networks operated and managed by several Italian institutions. The dataset includes 54,000 earthquakes that occurred between January 2005 and January 2020 in Italy and surrounding regions, with magnitude ranging from 0.0 to 6.5 (see <xref ref-type="bibr" rid="B31">Michelini et al., 2021</xref> for further details). Each datum consists of a 120&#xa0;s time window. Each waveform is associated with upward, downward, or undefined polarity. We excluded all those events with undefined polarity. In addition, selecting only the vertical component of velocimeters data, we achieved 161,198 seismic traces of which 103,530 showed upward polarity and 57,668 downward polarity. We will refer to these waveforms as <italic>dataset A</italic> (<xref ref-type="fig" rid="F1">Figure 1A</xref>). We split this dataset into three subsets, respectively used as.<list list-type="simple">
<list-item>
<p>&#x2022; Training set: 141,972 waveforms (88.0% of the total data) corresponding to 23,878 events shown as red circles in <xref ref-type="fig" rid="F1">Figure 1A</xref>;</p>
</list-item>
<list-item>
<p>&#x2022; Validation set: 10,243 waveforms (6.4% of the total data) corresponding to 2,275 events shown as orange circles in <xref ref-type="fig" rid="F1">Figure 1A</xref>;</p>
</list-item>
<list-item>
<p>&#x2022; Testing set: 8,983 waveforms (5.6% of the total data) corresponding to 2,398 events shown as blue circles in <xref ref-type="fig" rid="F1">Figure 1A</xref>.</p>
</list-item>
</list>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Localization of seismic events, shown with circles along the Italian peninsula. <bold>(A)</bold> The 28,551 events considered in <italic>dataset A</italic> (derived from the INSTANCE dataset). Waveforms belonging to events displayed by red circles are used as training data. The orange and blue boxes respectively contain events used to derive validation and test waveforms data. <bold>(B)</bold> The 842 events present in <italic>dataset B (derived from</italic> <xref ref-type="bibr" rid="B34">Napolitano et al., 2021b</xref>), located in Southern Italy, whose waveforms are used as a second test set.</p>
</caption>
<graphic xlink:href="feart-11-1223686-g001.tif"/>
</fig>
<p>The spatial selection was made to avoid correlations between waveforms in the different sets, following the approach proposed by <xref ref-type="bibr" rid="B52">Uchide (2020)</xref>. It is noteworthy that the validation set comprises earthquakes from the Etna volcano region (orange box in <xref ref-type="fig" rid="F1">Figure 1A</xref>).</p>
<p>Then, we collected the 870 earthquakes (M<sub>L</sub> 1.8&#x2013;5.0), recorded during the 2010&#x2013;2014 Pollino (Southern Italy) seismic sequence (<xref ref-type="fig" rid="F1">Figure 1B</xref>) by three seismic networks (<italic>Istituto Nazionale di Geofisica e Vulcanologia</italic> (INGV), <italic>Universit&#xe0; della Calabria</italic> (UniCal) and <italic>Deutsche GeoForschungsZentrum</italic> (GFZ)) (<xref ref-type="bibr" rid="B37">Passarelli et al., 2012</xref>; <xref ref-type="bibr" rid="B30">Margheriti et al., 2013</xref>) and located in the new 3D velocity model by <xref ref-type="bibr" rid="B34">Napolitano et al. (2021b)</xref>. From these events, we selected the vertical components of the waveforms sampled at 100&#xa0;Hz, registered by velocimeters and with clear P-wave polarity. We refer to this dataset as <italic>dataset B</italic>. It comprises 4,072 manually picked waveforms derived from 824 out of the original 870 events collected. We used <italic>dataset B</italic> as a second test set to evaluate the performance of the neural network on data from a specific Italian tectonic setting. To avoid any possible overlapping between <italic>dataset A</italic> and <italic>dataset B</italic>, we removed the 821 common waveforms in the former dataset.</p>
<p>In addition, we used seismic traces from the Southern California Seismic Network (<xref ref-type="bibr" rid="B48">Ross et al., 2018</xref>) and western Japan region (<xref ref-type="bibr" rid="B17">Hara et al., 2019</xref>) to evaluate the network&#x2019;s generalization ability on waveforms from completely different regions. For this purpose, we selected the 863,151 waveforms belonging to the 273,882 earthquakes registered at 682 stations from the SCSN dataset. This constitutes the part of the test set with definite polarity used in <xref ref-type="bibr" rid="B48">Ross et al. (2018)</xref>, whose magnitudes lie in the range [&#x2212;1.0,7.2]. Similarly, we used 3,930 waveforms (M<sub>L</sub> -1.3&#x2013;6.2) constituting a part of the test set sampled at 100&#xa0;Hz provided to us by <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref>. The waveforms from the western Japan region were registered by stations operated by the National Research Institute for Earth Science and Disaster Prevention (NIED), the National Institute of Advanced Industrial Science and Technology (AIST), the Japan Meteorological Agency (JMA), and Kyoto University (<xref ref-type="bibr" rid="B17">Hara et al., 2019</xref>).</p>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methods</title>
<sec id="s3-1">
<title>3.1 Data visualization with SOM and label noise</title>
<p>Before training the network on part of <italic>dataset A</italic>, the preliminary step of our analysis has been the implementation of a data visualization technique to investigate the waveforms. To this end, we applied the Self Organizing Maps (SOM, <xref ref-type="bibr" rid="B21">Kohonen, T., 2013</xref>). This unsupervised machine learning technique is highly efficient in reducing the dimensionality of large datasets, by leveraging the similarities between the data, to cluster and visualize them in a low-dimensional grid, while preserving their topological structure. In order to focus the SOM on the features of our interest, the map was given a representation of the data in feature space. We normalized the traces to unit variance and we focused our attention on time windows of 0.26&#xa0;s (26 samples), which include the 0.20&#xa0;s preceding the P-arrivals and the 0.05&#xa0;s after. We used 5 samples after the arrival, as they were enough to capture the entire first oscillation of the seismic wave in the case of higher frequency earthquakes, and enough to point out the trend of the oscillation in the case of lower frequency earthquakes. A lower value was not sufficient to capture the trend of oscillations in low-frequency events, whereas with higher values we observed that the analysis also focused on the second oscillations. We employed 20 samples before the arrival as they constituted the minimum number required to capture the essential characteristics of the noise trend in each scenario. Features provided to the SOM were extracted either by the Principal Component Analysis (<xref ref-type="bibr" rid="B4">Bishop and Nasrabadi, 2006</xref>), to which the normalized 0.26-seconds-long time windows were provided, and by evaluating averages of 0.16-seconds-long moving temporal windows. The first average was calculated over the time window starting from 0.19&#xa0;s before the P-arrival, and the subsequent 9 averages were calculated on shifted windows, moving forward by 0.01&#xa0;s each (1 sample), with the last time window covering the last 0.16&#xa0;s (from 0.10&#xa0;s before the arrival to 0.05&#xa0;s after). In total, we gave the SOM 16 features, namely, the first 6 principal components and 10 moving averages. We chose to consider the principal components up to the sixth because it was a fair trade-off between the number of dimensions taken into account and the explained variance. By using six components, we were able to achieve a 95% explained variance.</p>
<p>In our analysis, the map nodes were organized in a two-dimensional hexagonal 8 &#xd7; 8 grid (<xref ref-type="sec" rid="s11">Supplementary Figure S1B</xref> gives a representation of the grid). After the SOM training stage, we displayed the waveforms&#x2019; clusters on the map of nodes. Each single node represents a cluster that contains all those data whose distance in input space is smaller than the distance to all other nodes. <xref ref-type="sec" rid="s11">Supplementary Figure S1A,S2A,S3A</xref> show the mean value of the waveforms contained in each node and one-fifth of the waveforms falling in each of them, respectively using the total, upward, and downward first-motion polarity. The number of waveforms in each cluster is represented by the size of the hexagons in <xref ref-type="sec" rid="s11">Supplementary Figure S1B,S2B,S3B</xref>. We observe that the map places most of the waveform with downward polarity on the left side of the grid (<xref ref-type="sec" rid="s11">Supplementary Figure S3B</xref>), especially in the upper part, while the waveforms with upward polarities are mostly placed on the right side of the grid (<xref ref-type="sec" rid="s11">Supplementary Figure S2B</xref>), with the more populated nodes situated in the lower part. The net separation between the two parts provides a strong indication that, generally, the polarities are resolved in an unambiguous way. Nevertheless, a problem often encountered is that the polarities can be mistakenly labeled. To overcome such difficulty, we investigated the SOM results in more detail.</p>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> shows in each cell the weighted percentage of traces with upward polarities contained in it. Since the number of downward polarities is smaller than the upward one in <italic>dataset A</italic>, a weighted percentage is required for a robust analysis. Specifically, the value <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, showed in the cell relative to the node located in the <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th row and <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th column of the grid, is:<disp-formula id="e1_1">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>w</mml:mi>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1.1)</label>
</disp-formula>where <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are respectively the number of upward and downward waveforms assigned to the node <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. We notice the presence of some cells whose percentages of upward polarities are less than 1% or more than 99%. Considering the possibility of labeling errors in the dataset, we hypothesize that the high (low)-populated-upward cells represent nodes where all or most of the waveforms share the same polarity. Consequently, we suppose that the 458 outliers traces falling in those nodes (namely, the waveforms with an assigned polarity different from the majority) are likely to be mislabeled examples.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Heatmap relative to SOM nodes, showing the weighted percentages <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of upward waveforms laying in each node. We infer that the waveforms with assigned polarity different from the majority falling in dark blue and dark red cells (percentages less than 1% or more than 99%) are mislabeled data.</p>
</caption>
<graphic xlink:href="feart-11-1223686-g002.tif"/>
</fig>
<p>In fact, we manually checked that at least 100 of the 123 down-labeled traces, which fell in nodes with a weighted percentage of up-labeled data above 99%, had indeed an upward polarity. Analogously, at least 237 of the 336 up-labeled traces, located in nodes with more than 99% down-populated data, were clear waveforms with negative polarity. The remaining traces were mostly unclear waveforms, where extracting polarity information was a challenging task also for a human analyst. We do not exclude the presence of other mislabeled data (respect to the 337 found by the SOM visualization). A visual inspection of 1,000 randomly selected waveforms highlighted that approximately 8% of waveforms are affected by some problems, such as noisy arrival times or not reliable polarity information.</p>
<p>This level of noise is very common in real-world datasets, especially in the case of such large ones, where the ratio of corrupted labels can cover, in some cases, up to 40% of the entire dataset (<xref ref-type="bibr" rid="B51">Song et al., 2022</xref>). Although it may appear to have drastic consequences to use problematic data to train a classifier, numerous studies have demonstrated that, with appropriate precautions and depending on the nature of the encountered noise, deep learning can exhibit remarkable robustness (<xref ref-type="bibr" rid="B46">Rolnick et al., 2017</xref>; <xref ref-type="bibr" rid="B11">Drory et al., 2018</xref>). Furthermore, other works highlight that noise can also be useful to better generalize (<xref ref-type="bibr" rid="B9">Damian et al., 2021</xref>).</p>
<p>Subsequent investigations revealed that attempting to clean our dataset yielded no significant benefits. Specifically, a second SOM visualization technique, similar to the one previously described, has been applied. This analysis aimed to analyze upward and downward polarity waveforms separately and enabled us to remove from dataset A approximately 10,000 waveforms. We excluded all the waveforms that fell within SOM nodes where we determined the majority of the data to be ambiguous or where extracting polarity information was very challenging. These waveforms comprised elements from the training, validation, and test sets. <xref ref-type="sec" rid="s11">Supplementary Figure S4</xref> shows some of the excluded nodes. In <xref ref-type="sec" rid="s11">Supplementary Table S1</xref>, we compare the performance of the network trained on the original training set with the network trained on the cleaned training set, presenting the performance on both the cleaned test set and the original test set. Notably, we observed no significant differences in the performance of the two networks, when tested on the same test-set. Therefore, despite the presence of mislabelling in our dataset, we have chosen not to exclude any waveform, but rather, we aimed to design a network that can effectively handle and mitigate the effects of label noise, without the need for a preliminary selection of data points, which can result in information loss.</p>
</sec>
<sec id="s3-2">
<title>3.2 CFM architecture and preprocessing stage</title>
<p>The CFM network exclusively utilizes the vertical component of waveforms that have been sampled at a frequency of 100&#xa0;Hz whose polarity information is available. To ensure consistency of the input data, all waveforms are subjected to a standardized preprocessing stage. Specifically, we subtracted to each waveform the mean value of the noise, from 200 samples (2.0&#xa0;s) before the corresponding P arrival time to 5 samples before (in order to not include in the value of the mean some unbalanced oscillations due to the seismic phase). Subsequently, the initial wave portion is emphasized by setting a clipping threshold, in order not to neglect any of the smaller oscillations resulting from the signal (<xref ref-type="bibr" rid="B52">Uchide T., 2020</xref>). In this work, the threshold is different for each data point. To decide its value, the amplitude of the highest peak among those preceding the arrival time by at least 5 points was considered for each waveform. The threshold is equal to 20 times the value of this amplitude. Each seismogram is normalized to its respective threshold value. The portion of the signal exceeding this threshold is cut off. Previous studies did not highlight a specific filtering standard. <xref ref-type="bibr" rid="B52">Uchide (2020)</xref> used a high-pass filter at 1&#xa0;Hz, while <xref ref-type="bibr" rid="B48">Ross et al. (2018)</xref> applied a filter between 1 and 20&#xa0;Hz. On the other hand, <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref> and <xref ref-type="bibr" rid="B5">Chakraborty et al. (2022)</xref> avoided using any filter. <ext-link ext-link-type="uri" xlink:href="https://www.sciencedirect.com/topics/computer-science/convolutional-neural-network">CNN</ext-link> (and other deep networks) are known to work well on raw data (<xref ref-type="bibr" rid="B15">Goodfellow et al., 2016</xref>), since they learn features during training, in a hierarchical way, where initial layers acquire local features from data and the final layers extract global features representing high-level information. Considering these factors, we decided not to apply any frequency filters to our data.</p>
<p>We chose as our training set the part of dataset A outside the two boxes depicted in <xref ref-type="fig" rid="F1">Figure 1A</xref>. Waveforms were presented in time windows of 160 samples (1.60s, 0.79 preceding the P-arrival and 0.80 after), with the 80th sample corresponding to the declared P-arrival times. During the training stage, we presented to the network both waveforms and their corresponding labels. Specifically, we assigned to a generic waveform x the label <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if its label in the dataset was &#x201c;upward&#x201d; polarity; else <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. As previously stated, dataset A contains 103,530 upward and 57,668 downward polarity waveforms, resulting in an upward/downward ratio of 1.8. Similar level of unbalance is present in the selected data constituting our training set (on 141,972 total waveforms 91,563 showed upward polarity, while 50,409 showed downward polarity). A class imbalance may lead the network to prioritize the majority class, resulting in overlooking the characteristics of the minority class (<xref ref-type="bibr" rid="B55">Wang et al., 2016</xref>). For this reason, we balanced the training data applying a data augmentation technique (<xref ref-type="bibr" rid="B52">Uchide T., 2020</xref>; <xref ref-type="bibr" rid="B5">Chakraborty et al., 2022</xref>; <xref ref-type="bibr" rid="B12">Falanga et al., 2022</xref>) that allowed us to use a single data twice: the original trace and the corresponding flipped one, obtained by multiplying &#x2212;1 and assigning it the opposite polarity. As a result, our augmented training set doubled in size, comprising 283,944 waveforms, with half exhibiting upward polarity and the remaining half exhibiting downward polarity. We did not augment test or validation data.</p>
<p>
<xref ref-type="fig" rid="F3">Figure 3</xref> represents the Convolutional Neural Network architecture used in the present study. The network architecture is divided into two stages, the first of which is represented by the Convolutional layers. They provide a very efficient way to extract relevant features from grid-like data (<xref ref-type="bibr" rid="B15">Goodfellow et al., 2016</xref>), such as in the case of 1D time series (<xref ref-type="bibr" rid="B20">Kiranyaz et al., 2021</xref>) or 2D grids of pixel, i.e., images (<xref ref-type="bibr" rid="B23">Krizhevsky et al., 2017</xref>). The ReLU activation function is employed after each convolutional layer, owing to its well-known benefits in facilitating the training process (<xref ref-type="bibr" rid="B23">Krizhevsky et al., 2017</xref>). After three of the five Convolutional layers, a MaxPooling layer is added, which reduces the dimension of the input, preserving the most important features, and helps the network to gain translational invariance (<xref ref-type="bibr" rid="B15">Goodfellow et al., 2016</xref>). We also added Dropout layers, which are known to improve performance in case of training with noisy labels (<xref ref-type="bibr" rid="B49">Rusiecki, 2020</xref>), and prevent overfitting. In the second part of the network, the classification task is performed. The final layer&#x2019;s sigmoid, or logistic, activation function produces an output in the range [0, 1]. This choice allows the network output to be interpreted as the probability of an input vector to belong to one of the two investigated classes. We have used a threshold value of 0.5, above which we interpret data as having upward polarity and below which we interpret data as having downward polarity.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Architecture of the CFM, the deep Convolutional neural network for First Motion polarity classification used in this study. Numbers under each layer indicate its shape (i.e., number of channels x number of samples). ConvPool and ConvDrop indicate convolution with maxpooling and convolution with dropout, respectively. The values of K under each convolutional layer indicate the corresponding kernel size. The Flatten procedure (light blue arrow) only reshapes the previous layer in a one-dimensional array, without affecting any value.</p>
</caption>
<graphic xlink:href="feart-11-1223686-g003.tif"/>
</fig>
<p>We set the binary cross-entropy as the loss function to be minimized. To train the network, we used the Stochastic Gradient Descent (<xref ref-type="bibr" rid="B45">Robbins and Monro, 1951</xref>). SGD is one of the most simple and effective optimization methods widely used, and it can lead to better generalization performance compared to other more sophisticated methods. SGD is considered to play a central role in the observed generalization abilities of deep learning, since its stochasticity, resulting from the mini-batch sampling procedure, can provide a crucial implicit regularization effect (<xref ref-type="bibr" rid="B1">Ali et al., 2020</xref>). Moreover, the implicit regularization properties of SGD (<xref ref-type="bibr" rid="B9">Damian et al., 2021</xref>) are particularly useful when dealing with noisy data. We exploited the Stochastic Gradient Descend with the addition of Momentum. The default learning rate of <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:mn>0.01</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> shows good performances (multiple training with learning rate in the range <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mn>0.007</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.015</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> did not highlight substantial differences). We fixed the momentum parameter to be equal to <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mn>0.8</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and the batch size value equal to 512.</p>
<p>We set the maximum number of epochs to 100 and, to prevent overfitting, we implemented an early stopping technique that interrupts the training if there is no improvement in the validation loss for 7 consecutive epochs. Early stopping is also an effective implicit regularization technique, which has been observed to be surprisingly effective in preventing overfitting to mislabeled data, especially when used in combination with first-order optimization algorithms, such as SGD (<xref ref-type="bibr" rid="B26">Li et al., 2020</xref>).</p>
</sec>
</sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<p>CFM was trained on waveforms outside blue and orange boxes in <xref ref-type="fig" rid="F1">Figure 1A</xref>. The early stopping technique stopped the training at epoch number 20. We then evaluated the performance on the test set derived from <italic>dataset A</italic> (<xref ref-type="fig" rid="F1">Figure 1A</xref>, blue box) and on the <italic>dataset B</italic> (<xref ref-type="fig" rid="F1">Figure 1B</xref>), expressing it through confusion matrices (<xref ref-type="fig" rid="F4">Figures 4A,B</xref>, respectively), showing the number of samples labeled consistently with the dataset (top-left and bottom-right) or oppositely (top-right and bottom-left). From them, we computed the accuracies, defined as the number of correct predictions divided by the total ones. The network reached accuracies of 97.4% and 96.3%, respectively.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Confusion matrices for <italic>dataset A</italic> <bold>(A)</bold> and <italic>dataset B</italic> <bold>(B)</bold> test sets. The <italic>x</italic>-axis shows network prediction, while the <italic>y</italic>-axis reports the labels present in the dataset. The accuracies for <italic>dataset A</italic> and <italic>dataset B</italic> are approximately 97.4% and 96.3%, respectively.</p>
</caption>
<graphic xlink:href="feart-11-1223686-g004.tif"/>
</fig>
<p>To provide a measure of the network&#x2019;s reliability, we evaluated its behavior as the output varies on <italic>dataset A</italic> test set. A classifier is said to be &#x2018;well-calibrated&#x2019; when its output probability can be directly interpreted as a confidence level (<xref ref-type="bibr" rid="B10">Dawid, 1982</xref>). For instance, a well-calibrated classifier should classify the samples such that among the samples to which it gave a predicted probability close to 0.8, approximately 80% actually belong to the positive class, which in our case is represented by upward polarity. <xref ref-type="fig" rid="F5">Figure 5</xref> represents a reliability diagram of our network (<xref ref-type="bibr" rid="B36">Niculescu-Mizil and Caruana, 2005</xref>), which indicates how often data points assigned a certain forecast output probability interval actually exhibit upward polarity (assigned in the dataset). Mathematically, the value of the height of the rectangle belonging to the bin <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to the empirical probability:<disp-formula id="e1_2">
<mml:math id="m16">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x7c;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>:</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1.2)</label>
</disp-formula>where <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a generic data point, <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is its label, <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>F</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> the network out probability and <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mo>&#xb7;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> represents the cardinality of the ensemble. Although reliability diagrams can be helpful for visualizing calibration, having a scalar summary statistic of calibration is more practical. To this end, we calculated the Expected Calibration Error (<xref ref-type="bibr" rid="B16">Guo et al., 2017</xref>):<disp-formula id="e1_3">
<mml:math id="m21">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>M</mml:mi>
</mml:munderover>
</mml:mstyle>
</mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1.3)</label>
</disp-formula>where <inline-formula id="inf19">
<mml:math id="m22">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of predictions in bin m, n is the total number of data points, and acc (<inline-formula id="inf20">
<mml:math id="m23">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) and conf (<inline-formula id="inf21">
<mml:math id="m24">
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) are the accuracy and confidence of bin m, respectively. The ECE values range in the interval [0, 1], and the lower they are, the better the calibration of a model. We obtained an ECE value of 3.7% for our network. In general, ECE values depend on the specific task and dataset involved. For a general comparison, refer to <xref ref-type="bibr" rid="B16">Guo et al., 2017</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Reliability diagram of the network. Predictions made by the model are grouped into bins based on their predicted probabilities. The heights of the bars are the proportion of true positive cases within each bin. Green edges represent the average predicted probability of the bin, i.e., the optimal calibration. Numbers on each bar indicate the upward (red) and downward (blue) polarity waveforms laying in each bin.</p>
</caption>
<graphic xlink:href="feart-11-1223686-g005.tif"/>
</fig>
<sec id="s4-1">
<title>4.1 CFM robustness to false annotations</title>
<p>We remember the SOM analysis of <xref ref-type="sec" rid="s3-1">Section 3.1</xref> revealed the presence of 337 waveforms with false labels (located within the nodes highlighted in <xref ref-type="fig" rid="F2">Figure 2</xref>). Since the training set covers the majority of <italic>dataset A</italic>, the majority of these outlier waveforms (specifically 311) also belong to it. Despite the fact that the training algorithm forces the network output to match the assigned label, we found that 310 out of the 337 misclassified waveforms are assigned to the correct class by CFM. <xref ref-type="fig" rid="F6">Figure 6</xref> shows some examples of such waveforms we identified in <xref ref-type="sec" rid="s3-1">Section 3.1</xref> and for which the network predicts correct polarities. Given that the network successfully corrected 92% of the false labels, we consider this as evidence of its ability to be robust to overfitting erroneous labels.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Some seismic traces erroneously labeled by the analyst that we identified with the SOM data visualization in <xref ref-type="sec" rid="s3-1">Section 3.1</xref>. On the top of each subplot, we annotate the magnitude of the event (M) and the signal-to-noise ratio (SNR). P<sub>assigned</sub> and P<sub>predicted</sub> refer to the polarity assigned in the dataset and the prediction of the network (with the corresponding probability to belong to the predicted class in square brackets).</p>
</caption>
<graphic xlink:href="feart-11-1223686-g006.tif"/>
</fig>
</sec>
<sec id="s4-2">
<title>4.2 Dealing with uncertain arrival times</title>
<p>In this section, to check the robustness of the network to uncertainty in arrival times, we evaluated the performance of the network including artificial time shifts in arrival times. To this end, we shifted each time-window of <italic>dataset A</italic> test set by a constant value of T samples, with values of T in the range [-20,20]. A value of T &#x3d; 5, for example, indicates that the time window center is located 5 samples (0.05&#xa0;s) past the declared P-arrival. Red line in <xref ref-type="fig" rid="F7">Figure 7</xref> shows the behavior of the network (trained on centered time-windows) varying T as test sample-shift. We notice that, as expected, accuracy is highest when there is no shift. Accuracy rapidly declines, dropping to 50% when there is a shift of &#x2b;10 samples, indicating a significant degradation in performance.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>The performances of the CFM network on the test set after the two different training strategies. The blue and green lines refer to the trainings with a time-shift in the training set, with a maximum value N of 5 and 10 samples respectively. The red line shows the training without random time-shifts in the training set. Performance is shown as a function of the different shifts T in the test set. Dashed black lines refer to accuracy levels of 0.5 and 0.75.</p>
</caption>
<graphic xlink:href="feart-11-1223686-g007.tif"/>
</fig>
<p>Anyway, uncertainty in fixing the onset of P-wave is a trouble that often affects experimental data becoming much more difficult to manage for different reasons: poor signal-to-noise ratio, magnitude of the events decreases (small-energy/magnitude earthquake), recording stations installed in densely populated areas, complex medium properties, volcanic environment.</p>
<p>For this reason, we explored the possibility of giving the network the ability to deal with uncertainty in P-arrival times. Specifically, we developed an aimed augmentation strategy and performed a second training strategy, including a time-shift in the training set too. We used a time-shift augmentation procedure perturbing the centering of time windows contained in the training set, leaving the validation set unperturbed.</p>
<p>In particular, we selected 50% of the training waveforms and applied two independent uniform random time-shifts to each. The first time-shift was selected from the range [-N, &#x2212;1], and the second from [1, N]. The original waveform and the two shifted versions were then included in the training set. We conducted two training sessions, on two augmented training sets, with N values of 5 and 10, respectively. Evaluating performance on unperturbed <italic>dataset A</italic> test set (T&#x3d;0), we observe accuracy levels of 97.2% (in the case of N&#x3d;10) and of 97.3% (in the case of N&#x3d;5), which are slightly lower than the correspondent obtained by the model trained on unperturbed waveforms. However, as shown by the blue and green lines in <xref ref-type="fig" rid="F7">Figure 7</xref>, adding time-shifts to the training set can lead to an improvement in performance in the presence of uncertain arrival times. In particular, we observed a broader plateau where the accuracy remains above 92.4%, even when dealing with shifts of 10 samples, in the case of N&#x3d;10 (green line), and it takes 17 translation test samples to reduce the accuracy below 75%.</p>
</sec>
<sec id="s4-3">
<title>4.3 Model generalization ability</title>
<p>To evaluate the generalization ability of the CFM network, we checked the capability to generate accurate predictions on new datasets coming from completely different geographic regions (Southern California and western Japan regions), using recordings obtained by different seismic networks, and far from the region (Italy) on which the net was trained on.</p>
<p>We first utilized the SCSN test dataset provided by <xref ref-type="bibr" rid="B48">Ross et al. (2018)</xref>. We excluded waveforms without assigned polarity, resulting in 863,151 traces suitable for our purposes. The network achieves an accuracy of 98.4% for waveforms with SNR greater than or equal to 10, while the accuracy is 96.3% for waveforms with a SNR less than 10. The overall accuracy is 97.5%, comparable to the model trained by <xref ref-type="bibr" rid="B48">Ross et al. (2018)</xref> on the SCSN dataset (i.e. 95%). <xref ref-type="fig" rid="F8">Figure 8A</xref> shows the confusion matrix related to the network prevision on the SCSN dataset.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Confusion matrices for SCSN <bold>(A)</bold> and western Japan <bold>(B)</bold> test sets. The accuracies are approximately 97.5% and 91.5%, respectively. We recall that the performance on the western Japan test set refers to the different training using 150 as input waveforms.</p>
</caption>
<graphic xlink:href="feart-11-1223686-g008.tif"/>
</fig>
<p>We furthermore tested the performance on the test set provided by <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref>, using only the 3,930 waveforms sampled at 100&#xa0;Hz. We recall that CFM inputs are 160-sample waveforms, whereas the dataset we received contains 150-sample waveforms. Therefore, we have decided to conduct an additional training while keeping all the settings presented in the previous sections unchanged, except for the input shape, which we have adjusted to 150 samples to ensure compatibility. This additional training resulted in similar performances on both the <italic>dataset A</italic> and <italic>dataset B</italic> test sets when compared to the performance achieved with the 160-sample training. The predictions on the <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref> test set are presented in <xref ref-type="fig" rid="F8">Figure 8B</xref>, from which one can compute an accuracy value of about 91.5%, slightly lower than the 95.4% obtained by the model of <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref>.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>First-motion polarity determination can be a challenging task even for expert analysts, mainly when dealing with small-magnitude events, in both tectonic and volcanic environments. Deep learning neural networks have been widely applied in geophysics. Among many other applications, they have been used to detect first-motion polarities (<xref ref-type="bibr" rid="B48">Ross et al., 2018</xref>; <xref ref-type="bibr" rid="B17">Hara et al., 2019</xref>; <xref ref-type="bibr" rid="B52">Uchide, 2020</xref>; <xref ref-type="bibr" rid="B5">Chakraborty et al., 2022</xref>).</p>
<p>In this work, we developed the CFM network, a straightforward Convolutional Neural Network that can accurately identify the first-motion waveform polarity. Our results showed that CFM achieved a testing accuracy of 97.4% when applied to previously unseen traces. CFM also shows well generalization abilities, resulting in high accuracies on waveforms recorded from seismic networks located in completely different regions than those utilized to derive the training set (i.e., waveforms derived from the SCSN and western Japan test sets). For the SCSN test set, as noted in the previous works by <xref ref-type="bibr" rid="B48">Ross et al. (2018)</xref>; <xref ref-type="bibr" rid="B5">Chakraborty et al. (2022)</xref>, performance is better when dealing with waveforms that have a SNR greater than 10. Even if this is confirmed in our results, our network shows a gap in performance on different SNR of 2.1% when tested on SCSN test set, which is significantly lower than the 7.9% reported by <xref ref-type="bibr" rid="B5">Chakraborty et al. (2022)</xref>. For the western Japan region, the accuracy achieved by CFM on the <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref> test set, at 91.5%, is slightly lower than the accuracies obtained on the other test sets and the one reported by <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref> themselves. However, a manual analysis of all 333 misclassified waveforms revealed that the polarity assigned in the test set was correct only in 29 cases, while for 59 waveforms the polarity identified by the model was correct. Other waveforms either presented ambiguous or unextractable polarity (119 waveforms) or had a considerable error in arrival time, up to 35 samples (126 waveforms). <xref ref-type="sec" rid="s11">Supplementary Figure S5</xref> provides a representation of the various cases. These findings confirm that the instances where the network does not perform well are remarkably limited, and its inferior performance cannot be attributed to shortcomings.</p>
<p>We observed that the employed implicit regularization strategies prevented the network from overfitting mislabeled data, resulting in the network&#x2019;s ability to correct false labeling, even when the mislabeled waveforms are present in the training set. In line with previous studies (<xref ref-type="bibr" rid="B52">Uchide, 2020</xref>; <xref ref-type="bibr" rid="B5">Chakraborty et al., 2022</xref>), we demonstrated that implementing a time-shift augmentation procedure can lead to a decrease in performance when applied to unperturbed waveforms. However, unlike previous works, our additional training stages uncovered that an accurate augmentation procedure enables the handling of uncertainties in arrival times with only a minimal loss in performance on the unaltered data.</p>
<p>We also observe CFM exhibiting good calibration properties, which is critical for ensuring a high level of reliability in the model&#x2019;s outputs, although we did not carry out any explicit calibration processes (<xref ref-type="bibr" rid="B16">Guo et al., 2017</xref>). In addition, we observe (<xref ref-type="fig" rid="F5">Figure 5</xref>) that when the network works on waveforms with defined polarity, as in our case, the vast majority of outputs lie in the ranges [0, 0.1] for downward polarity and [0.9, 1] for upward polarity, resulting in high reliability. Due to its well-calibration properties, CFM is able to produce accurate probability estimates, enabling us to make informed decisions based on the output probability values. For example, a threshold can be introduced to determine when to accept or reject a prediction.</p>
<p>In conclusion, our study introduces the robust and highly adaptable CFM network that holds significant potential for determining the P-wave polarities. The generalization ability of the algorithm in producing accurate prediction on waveforms registered in regions different from those used to derive training data and its ability to rectify previously misclassified polarities are noteworthy contributions of this research. CFM key selling point lies in its capability to efficiently revise or validate large volumes of analyst-derived first-motion polarities in historic catalogs using a consistent method. It is important to note that the algorithm relies on phase arrival times and therefore cannot handle catalogs without this information. Although the application was presented on manually obtained picks, our findings suggest that the CFM network can easily be adapted downstream of the application of an automatic P-phase detection and labeling network, which is currently being worked on as a future development. This integration would enhance its adaptability and streamline the resolution of poorly-determined focal mechanisms in catalogs by quickly and robustly rectifying mislabeled first-motion polarities in databases. Overall, our research lays the foundation for further advancements in accurately characterizing tectonic and volcanic seismic events and improving our understanding of focal mechanisms.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The datasets used in this study are publicly available for download. The INSTANCE dataset can be accessed at the following link: <ext-link ext-link-type="uri" xlink:href="https://data.ingv.it/en/dataset/471#additional-metadata">https://data.ingv.it/en/dataset/471#additional-metadata</ext-link>. The SCSN dataset is accessible at the following link: <ext-link ext-link-type="uri" xlink:href="https://scedc.caltech.edu/data/deeplearning.html">https://scedc.caltech.edu/data/deeplearning.html</ext-link>. The CFM network and dataset B used in this research can be found in the GitHub repository: <ext-link ext-link-type="uri" xlink:href="https://github.com/Nemenick/CFM.git">https://github.com/Nemenick/CFM.git</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>OA, SS, and PC conceived the work. GM performed all the analysis. GM and SS developed the algorithm and implemented the code. FN and OA prepared the seismic catalog. GM, FN, SS, OA, and MF worked on draft manuscript preparation. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>PRIN-2017 MATISSE project, No. 20177EPPN2, funded by the Italian Ministry of Education and Research.</p>
</sec>
<ack>
<p>We thank Yukitoshi Fukahata for providing us with the dataset of <xref ref-type="bibr" rid="B17">Hara et al. (2019)</xref>.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/feart.2023.1223686/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/feart.2023.1223686/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ali</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dobriban</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>The implicit regularization of stochastic gradient flow for least squares</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>233</fpage>&#x2013;<lpage>244</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alvizuri</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tape</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Full moment tensors for small events (M w&#x3c; 3) at Uturuncu volcano, Bolivia</article-title>. <source>Geophys. J. Int.</source> <volume>206</volume> (<issue>3</issue>), <fpage>1761</fpage>&#x2013;<lpage>1783</lpage>. <pub-id pub-id-type="doi">10.1093/gji/ggw247</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aoki</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Earthquake focal mechanisms as a stress meter of active volcanoes</article-title>. <source>Geophys. Res. Lett.</source> <volume>49</volume> (<issue>19</issue>), <fpage>e2022GL100482</fpage>. <pub-id pub-id-type="doi">10.1029/2022GL100482</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bishop</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Nasrabadi</surname>
<given-names>N. M.</given-names>
</name>
</person-group> (<year>2006</year>). <source>Pattern recognition and machine learning</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chakraborty</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cartaya</surname>
<given-names>C. Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Faber</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>R&#xfc;mpker</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Stoecker</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>PolarCAP&#x2013;A deep learning approach for first motion polarity classification of earthquake waveforms</article-title>. <source>Artif. Intell. Geosciences</source> <volume>3</volume>, <fpage>46</fpage>&#x2013;<lpage>52</lpage>. <pub-id pub-id-type="doi">10.1016/j.aiig.2022.08.001</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Holland</surname>
<given-names>A. A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>PhasePApy: A robust pure Python package for automatic identification of seismic phases</article-title>. <source>Seismol. Res. Lett.</source> <volume>87</volume> (<issue>6</issue>), <fpage>1384</fpage>&#x2013;<lpage>1396</lpage>. <pub-id pub-id-type="doi">10.1785/0220160019</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chiang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dreger</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Ford</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Walter</surname>
<given-names>W. R.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Source characterization of underground explosions from combined regional moment tensor and first&#x2010;motion analysis</article-title>. <source>Bull. Seismol. Soc. Am.</source> <volume>104</volume> (<issue>4</issue>), <fpage>1587</fpage>&#x2013;<lpage>1600</lpage>. <pub-id pub-id-type="doi">10.1785/0120130228</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dahm</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Brandsd&#xf3;ttir</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Moment tensors of microearthquakes from the Eyjafjallaj&#xf6;kull volcano in South Iceland</article-title>. <source>Geophys. J. Int.</source> <volume>130</volume> (<issue>1</issue>), <fpage>183</fpage>&#x2013;<lpage>192</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-246X.1997.tb00997.x</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Damian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Label noise sgd provably prefers flat global minimizers</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>34</volume>, <fpage>27449</fpage>&#x2013;<lpage>27461</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2106.06530</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dawid</surname>
<given-names>A. P.</given-names>
</name>
</person-group> (<year>1982</year>). <article-title>The well-calibrated Bayesian</article-title>. <source>J. Am. Stat. Assoc.</source> <volume>77</volume> (<issue>379</issue>), <fpage>605</fpage>&#x2013;<lpage>610</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.1982.10477856</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Drory</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Avidan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Giryes</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2018</year>). <source>On the resistance of neural nets to label noise</source>. <comment>arXiv preprint arXiv:1803.11410, 2</comment>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Falanga</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>De Lauro</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Petrosino</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rincon-Yanez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Senatore</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Semantically enhanced IoT-oriented seismic event detection: An application to Colima and Vesuvius volcanoes</article-title>. <source>IEEE Internet Things J.</source> <volume>9</volume> (<issue>12</issue>), <fpage>9789</fpage>&#x2013;<lpage>9803</lpage>. <pub-id pub-id-type="doi">10.1109/JIOT.2022.3148786</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ford</surname>
<given-names>S. R.</given-names>
</name>
<name>
<surname>Dreger</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Walter</surname>
<given-names>W. R.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Identifying isotropic events using a regional moment tensor inversion</article-title>. <source>J. Geophys. Res. Solid Earth</source> <volume>114</volume> (<issue>B01306</issue>), <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1029/2008JB005743</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Goodfellow</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Courville</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Deep learning</source>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>MIT press</publisher-name>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pleiss</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Weinberger</surname>
<given-names>K. Q.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>July) on calibration of modern neural networks</article-title>,&#x201d; in <source>International conference on machine learning</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>1321</fpage>&#x2013;<lpage>1330</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hara</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fukahata</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Iio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>P-wave first-motion polarity determination of waveform data in Western Japan using deep learning</article-title>. <source>Earth Planets Space</source> <volume>71</volume> (<issue>1</issue>), <fpage>127</fpage>. <pub-id pub-id-type="doi">10.1186/s40623-019-1111-x</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hardebeck</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Shearer</surname>
<given-names>P. M.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>A new method for determining first-motion focal mechanisms</article-title>. <source>Bull. Seismol. Soc. Am.</source> <volume>92</volume> (<issue>6</issue>), <fpage>2264</fpage>&#x2013;<lpage>2276</lpage>. <pub-id pub-id-type="doi">10.1785/0120010200</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Judson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Thelen</surname>
<given-names>W. A.</given-names>
</name>
<name>
<surname>Greenfield</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>R. S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Focused seismicity triggered by flank instability on K&#x12b;lauea&#x27;s Southwest Rift Zone</article-title>. <source>J. Volcanol. Geotherm. Res.</source> <volume>353</volume>, <fpage>95</fpage>&#x2013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.1016/j.jvolgeores.2018.01.016</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kiranyaz</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Avci</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Abdeljaber</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Ince</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gabbouj</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Inman</surname>
<given-names>D. J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>1D convolutional neural networks and applications: A survey</article-title>. <source>Mech. Syst. Signal Process.</source> <volume>151</volume>, <fpage>107398</fpage>. <pub-id pub-id-type="doi">10.1016/j.ymssp.2020.107398</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kohonen</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Essentials of the self-organizing map</article-title>. <source>Neural Netw.</source> <volume>37</volume>, <fpage>52</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2012.09.018</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kong</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Walter</surname>
<given-names>W. R.</given-names>
</name>
<name>
<surname>Pyle</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Koper</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Schmandt</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Combining deep learning with physics based features in explosion&#x2010;earthquake discrimination</article-title>. <source>Geophys. Res. Lett.</source> <volume>49</volume> (<issue>13</issue>), <fpage>e2022GL098645</fpage>. <pub-id pub-id-type="doi">10.1029/2022GL098645</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krizhevsky</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Imagenet classification with deep convolutional neural networks</article-title>. <source>Commun. ACM</source> <volume>60</volume> (<issue>6</issue>), <fpage>84</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.1145/3065386</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>La Rocca</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Galluzzo</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Focal mechanisms of recent seismicity at Campi Flegrei, Italy</article-title>. <source>J. Volcanol. Geotherm. Res.</source> <volume>388</volume>, <fpage>106687</fpage>. <pub-id pub-id-type="doi">10.1016/j.jvolgeores.2019.106687</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source> <volume>521</volume> (<issue>7553</issue>), <fpage>436</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Soltanolkotabi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Oymak</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>June). Gradient descent with early stopping is provably robust to label noise for overparameterized neural networks</article-title>,&#x201d; in <source>International conference on artificial intelligence and statistics</source> (<publisher-loc>New York</publisher-loc>: <publisher-name>PMLR</publisher-name>), <fpage>4313</fpage>&#x2013;<lpage>4324</lpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>FocMech-flow: Automatic determination of P-wave first-motion polarity and focal mechanism inversion and application to the 2021 yangbi earthquake sequence</article-title>. <source>Appl. Sci.</source> <volume>13</volume> (<issue>4</issue>), <fpage>2233</fpage>. <pub-id pub-id-type="doi">10.3390/app13042233</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Linville</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pankow</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Draelos</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep learning models augment analyst decisions for event discrimination</article-title>. <source>Geophys. Res. Lett.</source> <volume>46</volume> (<issue>7</issue>), <fpage>3643</fpage>&#x2013;<lpage>3651</lpage>. <pub-id pub-id-type="doi">10.1029/2018GL081119</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>L&#xf3;pez-P&#xe9;rez</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Garc&#xed;a</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ben&#xed;tez</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Molina</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A contribution to deep learning approaches for automatic classification of volcanovolcano-seismic events: Deep Gaussian processes</article-title>. <source>IEEE Trans. Geosci. Remote Sens.</source> <volume>59</volume> (<issue>5</issue>), <fpage>3875</fpage>&#x2013;<lpage>3890</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2020.3022995</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Margheriti</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Amato</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Braun</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Cecere</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>D&#x27;Ambrosio</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>De Gori</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Emergenza nell&#x2019;area del Pollino: Le attivit&#xe0; della Rete sismica mobile</article-title>. <source>Rapp. Tec. INGV</source> <volume>252</volume>, <fpage>1</fpage>&#x2013;<lpage>40</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Michelini</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cianetti</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gaviano</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Giunchi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jozinovi&#x107;</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lauciani</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>INSTANCE&#x2013;the Italian seismic dataset for machine learning</article-title>. <source>Earth Syst. Sci. Data</source> <volume>13</volume> (<issue>12</issue>), <fpage>5509</fpage>&#x2013;<lpage>5544</lpage>. <pub-id pub-id-type="doi">10.5194/essd-13-5509-2021</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miller</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Julian</surname>
<given-names>B. R.</given-names>
</name>
<name>
<surname>Foulger</surname>
<given-names>G. R.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Three-dimensional seismic structure and moment tensors of non-double-couple earthquakes at the Hengill&#x2013;Grensdalur volcanic complex, Iceland</article-title>. <source>Geophys. J. Int.</source> <volume>133</volume> (<issue>2</issue>), <fpage>309</fpage>&#x2013;<lpage>325</lpage>. <pub-id pub-id-type="doi">10.1046/j.1365-246X.1998.00492.x</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mousavi</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Beroza</surname>
<given-names>G. C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep-learning seismology</article-title>. <source>Science</source> <volume>377</volume> (<issue>6607</issue>), <fpage>eabm4470</fpage>. <pub-id pub-id-type="doi">10.1126/science.abm4470</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Napolitano</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Amoroso</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>La Rocca</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gervasi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gabrielli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Capuano</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021b</year>). <article-title>Crustal structure of the seismogenic volume of the 2010&#x2013;2014 Pollino (Italy) seismic sequence from 3D P-and S-wave tomographic images</article-title>. <source>Front. Earth Sci.</source> <volume>9</volume>, <fpage>735340</fpage>. <pub-id pub-id-type="doi">10.3389/feart.2021.735340</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Napolitano</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Galluzzo</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gervasi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Scarpa</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>La Rocca</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2021a</year>). <article-title>Fault imaging at Mt Pollino (Italy) from relative location of microearthquakes</article-title>. <source>Geophys. J. Int.</source> <volume>224</volume> (<issue>1</issue>), <fpage>637</fpage>&#x2013;<lpage>648</lpage>. <pub-id pub-id-type="doi">10.1093/gji/ggaa407</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Niculescu-Mizil</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Caruana</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Predicting good probabilities with supervised learning</article-title>,&#x201d; in <conf-name>Proceedings of the 22nd international conference on Machine learning</conf-name>, <conf-loc>New York</conf-loc>, <conf-date>August 2005</conf-date>, <fpage>625</fpage>&#x2013;<lpage>632</lpage>. <pub-id pub-id-type="doi">10.1145/1102351.1102430</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Passarelli</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Roessler</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Aladino</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Maccaferri</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Moretti</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lucente</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <source>Pollino seismic experiment (2012-2014)</source>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Perol</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gharbi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Denolle</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Convolutional neural network for earthquake detection and location</article-title>. <source>Sci. Adv.</source> <volume>4</volume> (<issue>2</issue>), <fpage>e1700578</fpage>. <pub-id pub-id-type="doi">10.1126/sciadv.1700578</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pesicek</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Sileny</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Prejean</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Thurber</surname>
<given-names>C. H.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Determination and uncertainty of moment tensors for microearthquakes at Okmok Volcano, Alaska</article-title>. <source>Geophys. J. Int.</source> <volume>190</volume> (<issue>3</issue>), <fpage>1689</fpage>&#x2013;<lpage>1709</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-246X.2012.05574.x</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prezioso</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Piccialli</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Convertito</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A data-driven artificial neural network model for the prediction of ground motion from induced seismicity: The case of the Geysers geothermal field</article-title>. <source>Front. Earth Sci.</source> <volume>10</volume>, <fpage>2193</fpage>. <pub-id pub-id-type="doi">10.3389/feart.2022.917608</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pugh</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Christie</surname>
<given-names>P. A. F.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Automatic Bayesian polarity determination</article-title>. <source>Geophys. J. Int.</source> <volume>206</volume> (<issue>1</issue>), <fpage>275</fpage>&#x2013;<lpage>291</lpage>. <pub-id pub-id-type="doi">10.1093/gji/ggw146</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reasenberg</surname>
<given-names>P. A.</given-names>
</name>
</person-group> (<year>1985</year>). <article-title>FPFIT, FPPLOT, and FPPAGE: Fortran computer programs for calculating and displaying earthquake fault-plane solutions</article-title>. <source>U. S. Geol. Surv. Open-File Rep.</source>, <fpage>85</fpage>&#x2013;<lpage>739</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Richardson</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Feller</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Seismic data denoising and deblending using deep learning</source>. <comment>arXiv preprint arXiv:1907.01497</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.1907.01497</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rincon-Yanez</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>De Lauro</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Petrosino</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Senatore</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Falanga</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Identifying the fingerprint of a volcano in the background seismic noise from machine learning-based approach</article-title>. <source>Appl. Sci.</source> <volume>12</volume> (<issue>14</issue>), <fpage>6835</fpage>. <pub-id pub-id-type="doi">10.3390/app12146835</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Robbins</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Monro</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>1951</year>). <article-title>A stochastic approximation method</article-title>. <source>Ann. Math. Stat.</source> <volume>22</volume>, <fpage>400</fpage>&#x2013;<lpage>407</lpage>. <pub-id pub-id-type="doi">10.1214/aoms/1177729586</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Rolnick</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Veit</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Belongie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Shavit</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Deep learning is robust to massive label noise</source>. <comment>arXiv preprint arXiv:1705.10694</comment>. <pub-id pub-id-type="doi">10.48550/arXiv.1705.10694</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roman</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Neuberg</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Luckett</surname>
<given-names>R. R.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Assessing the likelihood of volcanic eruption through analysis of volcanotectonic earthquake fault&#x2013;plane solutions</article-title>. <source>Earth Planet. Sci. Lett.</source> <volume>248</volume> (<issue>1-2</issue>), <fpage>244</fpage>&#x2013;<lpage>252</lpage>. <pub-id pub-id-type="doi">10.1016/j.epsl.2006.05.029</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ross</surname>
<given-names>Z. E.</given-names>
</name>
<name>
<surname>Meier</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Hauksson</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>P wave arrival picking and first&#x2010;motion polarity determination with deep learning</article-title>. <source>
<italic>J. Geophys. Res.</italic> Solid Earth</source> <volume>123</volume> (<issue>6</issue>), <fpage>5120</fpage>&#x2013;<lpage>5129</lpage>. <pub-id pub-id-type="doi">10.1029/2017JB015251</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Rusiecki</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Standard dropout as remedy for training deep neural networks with label noise</article-title>,&#x201d; in <source>Theory and applications of dependable computer systems: Proceedings of the fifteenth international conference on dependability of computer systems DepCoS-RELCOMEX, june 29&#x2013;july 3, 2020, brun&#xf3;w, Poland</source> (<publisher-loc>Germany</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), <fpage>534</fpage>&#x2013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-48256-5_52</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Snoke</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>W. H. K.</given-names>
</name>
<name>
<surname>Kanamori</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Jennings</surname>
<given-names>P. C.</given-names>
</name>
<name>
<surname>Kisslinger</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Focmec: Focal mechanism determinations</article-title>. <source>Int. Handb. Earthq. Eng. Seismol.</source> <volume>85</volume>, <fpage>1629</fpage>&#x2013;<lpage>1630</lpage>. <pub-id pub-id-type="doi">10.1016/S0074-6142(03)80291-7</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Shin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J. G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Learning from noisy labels with deep neural networks: A survey</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source>, <fpage>1</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2022.3152527</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uchide</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Focal mechanisms of small earthquakes beneath the Japanese islands based on first-motion polarities picked using deep learning</article-title>. <source>Geophys. J. Int.</source> <volume>223</volume>, <fpage>1658</fpage>&#x2013;<lpage>1671</lpage>. <pub-id pub-id-type="doi">10.1093/gji/ggaa401</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uchide</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Shiina</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Imanishi</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Stress map of Japan: Detailed nationwide crustal stress field inferred from focal mechanism solutions of numerous microearthquakes</article-title>. <source>J. Geophys. Res.</source> <volume>127</volume> (<issue>6</issue>), <fpage>e2022JB024036</fpage>. <pub-id pub-id-type="doi">10.1029/2022JB024036</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vavry&#x10d;uk</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Iterative joint inversion for stress and fault orientations from focal mechanisms</article-title>. <source>Geophys. J. Int.</source> <volume>199</volume> (<issue>1</issue>), <fpage>69</fpage>&#x2013;<lpage>77</lpage>. <pub-id pub-id-type="doi">10.1093/gji/ggu224</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Kennedy</surname>
<given-names>P. J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Training deep neural networks on imbalanced data sets</article-title>,&#x201d; in <conf-name>2016 international joint conference on neural networks (IJCNN)</conf-name>, <conf-loc>Vancouver</conf-loc>, <conf-date>24-29 July 2016</conf-date> (<publisher-name>IEEE</publisher-name>), <fpage>4368</fpage>&#x2013;<lpage>4374</lpage>. <pub-id pub-id-type="doi">10.1109/IJCNN.2016.7727770</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>W&#xe9;ber</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Probabilistic joint inversion of waveforms and polarity data for double-couple focal mechanisms of local earthquakes</article-title>. <source>Geophys. J. Int.</source> <volume>213</volume> (<issue>3</issue>), <fpage>1586</fpage>&#x2013;<lpage>1598</lpage>. <pub-id pub-id-type="doi">10.1093/gji/ggy096</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Roman</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Le M&#xe9;vel</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Power</surname>
<given-names>J. A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Earthquakes indicated stress field change during the 2006 unrest of Augustine Volcano, Alaska</article-title>. <source>Geophys. Res. Lett.</source> <volume>49</volume>, <fpage>e2022GL097958</fpage>. <pub-id pub-id-type="doi">10.1029/2022gl097958</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>L. S.</given-names>
</name>
<name>
<surname>Helmberger</surname>
<given-names>D. V.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Source estimation from broadband regional seismograms</article-title>. <source>Bull. Seismol. Soc. Am.</source> <volume>84</volume> (<issue>1</issue>), <fpage>91</fpage>&#x2013;<lpage>104</lpage>.</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>DiTingMotion: A deep-learning first-motion-polarity classifier and its application to focal mechanism inversion</article-title>. <source>Front. Earth Sci.</source> <volume>11</volume>, <fpage>335</fpage>. <pub-id pub-id-type="doi">10.3389/feart.2023.1103914</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Beroza</surname>
<given-names>G. C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>PhaseNet: A deep-neural-network-based seismic arrival-time picking method</article-title>. <source>Geophys. J. Int.</source> <volume>216</volume> (<issue>1</issue>), <fpage>261</fpage>&#x2013;<lpage>273</lpage>. <pub-id pub-id-type="doi">10.1093/gji/ggy423</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>