<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1610163</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Toward non-invasive early pest surveillance: cross-modal adaptation using PLMS acoustic-visual representation and pre-trained transfer learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wu</surname>
<given-names>Yaqin</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3028501/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cheng</surname>
<given-names>Lijun</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hao</surname>
<given-names>Wangli</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3147285/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Niu</surname>
<given-names>Jianjun</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Yuze</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chang</surname>
<given-names>Yan</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lv</surname>
<given-names>Jia</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Xuru</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>School of Software, Shanxi Agricultural University</institution>, <addr-line>Jinzhong, Shanxi</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1257877/overview">Jian Lian</ext-link>, Shandong Management University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1132679/overview">Phumudzo Patrick Tshikhudo</ext-link>, University of South Africa, South Africa</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1763479/overview">Aibin Chen</ext-link>, Central South University Forestry and Technology, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Yaqin Wu, <email xlink:href="mailto:wyq0902@sxau.edu.cn">wyq0902@sxau.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>10</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1610163</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wu, Cheng, Hao, Niu, Li, Chang, Lv and Li.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wu, Cheng, Hao, Niu, Li, Chang, Lv and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Pest infestations pose significant threats to agricultural productivity and ecological balance, making early prevention crucial for effective management. Toward non-invasive early-stage pest surveillance, this study introduces a novel cross-modal adaptation paradigm, leveraging the comprehensive bioacoustic repository, InsectSound1000 database. Firstly, the methodology initiates with adaptive audio preprocessing, where raw signals are filtered using the low-pass filter to remove high-frequency interference, followed by the downsampling operation to prevent aliasing and reduce computational complexity. Secondly, Patch-level log-scale mel spectrum (PLMS) spectrograms are proposed to convert acoustic signals into visual representations, refining time-frequency patterns through patch-level hierarchical decomposition to capture low-frequency and localized spectral features. The logarithmic transformation further enhances subtle low-frequency insect sound characteristics, optimizing feature analysis and boosting model sensitivity and generalization. Next, the PLMS acoustic-visual spectrograms undergo data augmentation prior to being processed by the pre-trained You Only Look Once version 11(YOLOv11) model for deep transfer learning, facilitating the efficient extraction of high-level semantic features. Finally, we compare the proposed algorithm with traditional acoustic features and networks, investigating how to balance preserving the frequency content of the signal and meeting computational requirements through optimized downsampling. Experimental results demonstrate that the proposed method achieves an Accuracy@1 of 96.49%, a Macro-F1 score of 96.49%, and a Macro-AUC of 99.93% at the 2500Hz sampling rate, showcasing its superior performance. These findings indicate that cross-modal adaptation with PLMS spectrograms and YOLOv11-based transfer learning can significantly enhance pest sound detection, providing a robust framework for non-invasive, early-stage agricultural pest surveillance.</p>
</abstract>
<kwd-group>
<kwd>pest surveillance</kwd>
<kwd>insectsound1000</kwd>
<kwd>acoustic-visual</kwd>
<kwd>PLMS</kwd>
<kwd>transfer learning</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="9"/>
<equation-count count="16"/>
<ref-count count="52"/>
<page-count count="25"/>
<word-count count="13748"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Agriculture plays a critical role in the global economy, food security, and rural development. As the backbone of many nations, particularly in developing regions, agriculture supports the livelihoods of billions of people and is responsible for producing food, fiber, and raw materials that sustain both local and global markets (<xref ref-type="bibr" rid="B14">FAO, 2024</xref>). Meanwhile, the increasing significance and attention towards sustainable agriculture as a solution to global challenges and a driver of rural development (<xref ref-type="bibr" rid="B39">Rusdiyana et&#xa0;al., 2024</xref>). However, the sector faces significant challenges, especially due to climate change, population growth, pest invasion and the increasing demand for sustainability. Climate change, in particular, has a significant negative impact on agricultural productivity (<xref ref-type="bibr" rid="B2">Bai et&#xa0;al., 2024</xref>). Concurrently, the rapidly growing global population intensifies concerns over food security. By 2050, the world will need to feed approximately 10 billion people, without depleting the planet&#x2019;s resources or damaging the environment (<xref ref-type="bibr" rid="B17">Ghosh et&#xa0;al., 2024</xref>). Consequently, ensuring food security while preserving environmental sustainability has emerged as one of the most pressing challenges of the 21st century (<xref ref-type="bibr" rid="B47">Varzakas and Smaoui, 2024</xref>). Moreover, the escalating frequency and severity of pest invasions, frequently exacerbated by climate variability and global trade, further jeopardize crop yields and compromise the stability of agricultural systems worldwide. Among these challenges, pest infestation is a critical concern, as it directly impacts crop yields, food security, and the livelihoods of millions of farmers. To address this, Integrated Pest Management has emerged as a sustainable solution, effectively minimizing reliance on pesticides while simultaneously improving crop productivity and promoting ecosystem health (<xref ref-type="bibr" rid="B52">Zhou et&#xa0;al., 2024</xref>).</p>
<p>Pest infestation is critical to sustaining agricultural productivity. Certain insects, rodents, and other pests inflict significant damage to crops, resulting in major economic losses and jeopardizing food security. In addition to direct agricultural impacts, pest infestations can have profound environmental and social consequences. Environmentally, pests may lower biodiversity, endanger local species through predation and competition, destroy habitats, and interfere with pollination and other ecological processes. Socially, pests threaten food security, which can result in starvation and social unrest, as well as harming urban surroundings and cultural heritage, like when invasive pests cause urban trees to disappear. Furthermore, the health and well-being of communities can also be impacted by pests; urban pests like bedbugs and rats frequently indicate underlying psychological, social, or economic problems in local communities. Effective pest management strategies aim to promote sustainable agricultural practices, significantly reduce reliance on synthetic pesticides, and address a range of socio-economic, environmental, and human health challenges (<xref ref-type="bibr" rid="B7">Deguine et&#xa0;al., 2021</xref>). Although traditional pest control methods, such as chemical pesticides, have been widely adopted, their long-term adverse effects, including pesticide resistance, environmental degradation, and harm to non-target species, are becoming increasingly apparent (<xref ref-type="bibr" rid="B16">Gandara et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B27">Liu et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B28">Liu et&#xa0;al., 2024</xref>). Consequently, there is an urgent need for more sustainable and efficient pest control alternatives.</p>
<p>Among the various methods developed, image-based pest detection and control systems have garnered significant attention, driven by recent advancements in computer vision and machine learning techniques. For example, <xref ref-type="bibr" rid="B26">Li et&#xa0;al. (2021)</xref> explores the technical methods and frameworks of deep learning for smart pest monitoring, focusing on insect pest classification and detection based on field images. The study provides a comprehensive analysis of methodologies across key stages, including image acquisition, data preprocessing, and modeling techniques. By analyzing the captured images, the system could accurately identify the presence and quantity of pests, enabling farmers to take appropriate control measures promptly. However, image-based methods also face inherent challenges, including the lack of large, well-annotated image datasets, the impact of environmental factors on insect features, and difficulties in detecting hidden or camouflaged pests arising from their position and similarity to other species (<xref ref-type="bibr" rid="B33">Ngugi et&#xa0;al., 2021</xref>). Additionally, these systems may require substantial computational resources and be sensitive to environmental factors such as lighting and weather conditions, all of which complicate AI-based approaches (<xref ref-type="bibr" rid="B23">Kiobia et&#xa0;al., 2023</xref>).</p>
<p>Given these limitations, researchers have been exploring alternative and complementary techniques, one of which is pest detection through acoustic recognition technology. Unlike image-based methods that rely on visual cues, acoustic-based detection utilizes the unique acoustic features emitted by pests during behaviors such as feeding, mating, or movement, offering a promising avenue for accurate identification and monitoring. For example, the chirping of <italic>crickets</italic> or the buzzing of certain <italic>beetles</italic> can be distinct identifiers. Moreover, acoustic technology offers valuable insights into stored insect behavior, physiology, abundance, and distribution, providing information that is otherwise challenging to obtain through traditional methods (<xref ref-type="bibr" rid="B30">Mankin et&#xa0;al., 2021</xref>). To build upon this, acoustic technology can operate effectively in complete darkness or in environments with dense foliage, where visual access is restricted. Unlike visual methods, it is unaffected by variations in lighting that may degrade image quality. Moreover, the hardware required for sound acquisition, such as basic microphones, is often more cost-effective than high-resolution cameras. However, acoustic detection can be susceptible to environmental noise and adverse weather conditions such as wind, rain, or foliage movement, which may mask or distort target acoustic signals. Following this line of research, a low-cost real-time platform for the acoustic detection of <italic>cicadas</italic> in plantations was introduced (<xref ref-type="bibr" rid="B11">Escola et&#xa0;al., 2020</xref>). Similarly, the system proposed (<xref ref-type="bibr" rid="B1">Ali et&#xa0;al., 2024</xref>), driven by Internet of Things-based (IoT-based) computerized components, utilized machine learning on insect acoustic recordings, further enhancing the accuracy and reliability of pest detection. Thus, acoustic-based pest detection provides an effective, cost-efficient, and adaptable solution for pest monitoring across diverse environments.</p>
<p>In this study, a novel approach for early prevention of pest infestation is proposed, utilizing a cross-modal adaptation framework based on the InsectSound1000 database. The main contributions of the proposed approach are summarized as follows:</p>
<list list-type="roman-lower">
<list-item>
<p>We introduced an advanced methodology for transforming one-dimensional time-series signals into two-dimensional PLMS spectrograms, facilitating a more structured and informative acoustic-visual representation of insect acoustic characteristics.</p>
</list-item>
<list-item>
<p>We leveraged transfer learning by fine-tuning the pre-trained YOLOv11 model, capitalizing on its robust feature extraction capabilities acquired from large-scale datasets after data Augmentation.</p>
</list-item>
<list-item>
<p>We optimized computational efficiency by fine-tuning only a subset of model parameters, minimizing Floating Point Operations Per Second (FLOPS) and parameter count to achieve the lowest resource footprint among compared models, enabling real-time pest surveillance.</p>
</list-item>
<list-item>
<p>We explored the impact of different patch sizes and sample rates on classification performance, investigating the optimal parameter tuning to balance preserving the frequency content of the signal while meeting computational requirements.</p>
</list-item>
<list-item>
<p>Extensive experiments were conducted using the InsectSound1000 dataset. Results demonstrate that our method achieves superior classification performance, significantly outperforming existing related works.</p>
</list-item>
</list>
</sec>
<sec id="s2">
<label>2</label>
<title>Related works</title>
<p>A substantial cohort of researchers conducts investigations on acoustic-based insect detection. As shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, this section provides a comprehensive synthesis of the relevant literature in terms of methodology, dataset, strengths and limitation to provide context for the present study.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>A concise overview of the literature reviews.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Methodology</th>
<th valign="middle" align="left">Dataset</th>
<th valign="middle" align="left">Strengths</th>
<th valign="middle" align="left">Limitations</th>
<th valign="middle" align="left">Ref.</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Transformer-based networks with data augmentation</td>
<td valign="middle" align="left">518 audio samples of 15 <italic>bee</italic> species</td>
<td valign="middle" align="left">F1: 64.5%, Accuracy: 82.2%</td>
<td valign="middle" align="left">Small dataset; Class imbalance; Reliance on pre-training</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B15">Ferreira et&#xa0;al., 2025</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">DFSM with Efficientnet and dual towers</td>
<td valign="middle" align="left">InsectSet32</td>
<td valign="middle" align="left">Accuracy: 80.26%, outperforms SOTA by 3%</td>
<td valign="middle" align="left">Class imbalance; Insufficient data; Unaddressed real-world issues (insect sound variability and scalability for large-scale pest monitoring)</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B19">He et&#xa0;al., 2024</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">ML algorithms with MFCC features and data augmentation</td>
<td valign="middle" align="left">Sound recordings of <italic>cicada, beetle, termite</italic>, and <italic>cricket</italic>
</td>
<td valign="middle" align="left">Improved generalization, reduced overfitting, diverse data augmentation</td>
<td valign="middle" align="left">Potential over-reliance on augmentation; Limited to four insect types</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B49">Wang and Vhaduri, 2024</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">MEMS microphone, multi-layer CNN</td>
<td valign="middle" align="left">Sounds of <italic>lesser grain borer, rice weevil, and red flour beetle</italic> in stored paddy grains</td>
<td valign="middle" align="left">Accuracy: 84.51%, non-chemical pest detection, high information density handling</td>
<td valign="middle" align="left">Limited to adult insect stages, potential noise interference in storage environments; Lack of diversity</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B3">Balingbing et&#xa0;al., 2024</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">Hyper-parameter tuning, MFCC</td>
<td valign="middle" align="left">BUZZ1, BUZZ2, and add_BUZZ2</td>
<td valign="middle" align="left">Accuracy: 96.9%</td>
<td valign="middle" align="left">Limited to <italic>bee</italic> buzzing recognition</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B36">Phan et&#xa0;al., 2023</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">Sound-to-image conversion, feature fusion, DL classification</td>
<td valign="middle" align="left">Recordings from date palm trees in Al-Ahssa, Saudi Arabia</td>
<td valign="middle" align="left">Outperformed existing techniques for public datasets</td>
<td valign="middle" align="left">Limited to <italic>red palm weevil</italic> infestation classification</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B5">Boulila et&#xa0;al., 2023</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">UAV visual-acoustic system, DL source separation, spectral denoising, CNN transfer learning</td>
<td valign="middle" align="left">100+in-field 48 Megapixels(MP) photos, 16 species audio recordings</td>
<td valign="middle" align="left">Precision(visual: 0.92, acoustic: 0.87), Recall (visual: 0.84, acoustic: 0.90), cost-effective (&lt;$1000/unit), covers 12,500 m&#xb2;/hr</td>
<td valign="middle" align="left">Limited to <italic>grasshoppers</italic>; Requires UAV deployment; Noise interference challenges</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B50">Zhang, 2023</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">MFCC features, CNN</td>
<td valign="middle" align="left">2800 acoustic samples</td>
<td valign="middle" align="left">Precision (positive: 0.89, negative: 0.98), Recall (positive: 0.98, negative: 0.90), F1 (positive:0.93, negative: 0.94)</td>
<td valign="middle" align="left">Limited to <italic>Rice Weevils</italic>; Requires high-performance microphone</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B32">Montemayor et&#xa0;al., 2024</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">IoT and DMF-ResNet</td>
<td valign="middle" align="left">Bug Bytes sound library</td>
<td valign="middle" align="left">Accuracy (99.75%), Precision (99.18%), Recall (99.08%), F1 score (99.11%)</td>
<td valign="middle" align="left">High initial cost; Limited to specific pests</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B9">Dhanaraj et&#xa0;al., 2024</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">LEAF and mel-spectrogram</td>
<td valign="middle" align="left">InsectSet32 (32 species), InsectSet47 (47species), InsectSet66 (66 species); Focused on <italic>Orthoptera</italic> and <italic>Cicadidae</italic>
</td>
<td valign="middle" align="left">InsectSet32: LEAF Accuracy (78%);<break/>InsectSet47: LEAF Accuracy (86%); InsectSet66: LEAF Accuracy (83%)</td>
<td valign="middle" align="left">Small dataset; Limited audio quality</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B13">Fai&#xdf; and Stowell, 2023</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">CNN-GRU model with Mel Spectrogram and Bayesian Optimization</td>
<td valign="middle" align="left">BUZZ1, BUZZ2, and add_BUZZ2</td>
<td valign="middle" align="left">Outperforms existing models by 1% in bee sound identification.</td>
<td valign="middle" align="left">Limited to <italic>bee</italic> buzzing sounds; Small improvement margin (1%).</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B46">Truong et&#xa0;al., 2023</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">ResNet-9</td>
<td valign="middle" align="left">Wingbeats; Fruitfiles; Abuzz</td>
<td valign="middle" align="left">High accuracy, reduced trainable parameters (90% reduction)</td>
<td valign="middle" align="left">Limited to <italic>fruit flies</italic> and <italic>mosquitoes</italic>
</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B43">Szekeres et&#xa0;al., 2023</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">Improved MFCC scanning with ML models</td>
<td valign="middle" align="left">Collected from MobCup, Quick Sounds, and Pixabay; includes 9 insect species</td>
<td valign="middle" align="left">Achieved 85.4% accuracy with kNN</td>
<td valign="middle" align="left">Limited to 9 insect species</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B4">Basak et&#xa0;al., 2022</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">Empirical Mode Decomposition (EMD) and Paraconsistent Feature Engineering (PFE) for feature extraction, SVM for classification</td>
<td valign="middle" align="left">1366 audio files (683 <italic>cicada</italic>, 683 noise) from S&#xe3;o Paulo and Minas Gerais, Brazil</td>
<td valign="middle" align="left">Accuracy (98%)</td>
<td valign="middle" align="left">Limited to <italic>Quesada gigas</italic> species</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B8">de Souza et&#xa0;al., 2022</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">Syllable segmentation, Spectrogram representation, CNN</td>
<td valign="middle" align="left">43 sound recordings of three <italic>cicada</italic> species</td>
<td valign="middle" align="left">Accuracy (66.67% to 100%), robust species recognition</td>
<td valign="middle" align="left">Small dataset; limited to <italic>cicada</italic> species; Dependent on syllable segmentation</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B45">Tey et&#xa0;al., 2022</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">IoT-based with fine-tuned InceptionResNet-V2</td>
<td valign="middle" align="left">TreeVibes database (1754 samples: 1023 clean, 731 infested)</td>
<td valign="middle" align="left">Accuracy (97.18%), effective transfer learning, real-time detection</td>
<td valign="middle" align="left">Limited to <italic>Red Palm Weevils</italic>; Small dataset</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B22">Karar et&#xa0;al., 2021</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">MFCC feature, CNN</td>
<td valign="middle" align="left">Insect sound library from ARS Center</td>
<td valign="middle" align="left">Accuracy (92.56%)</td>
<td valign="middle" align="left">Limited dataset</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B51">Zhang et&#xa0;al., 2021</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">MFCC and LFCC</td>
<td valign="middle" align="left">343 species of <italic>katydids, crickets and cicadas</italic>
</td>
<td valign="middle" align="left">Accuracy (98.07%)</td>
<td valign="middle" align="left">Limited to 3 species</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B35">Noda et&#xa0;al., 2019</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">Enhanced spectrogram, CNN</td>
<td valign="middle" align="left">47 types of insect sounds from USDA library</td>
<td valign="middle" align="left">Accuracy (97.87%), reduced data size, and faster training</td>
<td valign="middle" align="left">Small dataset</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B10">Dong et&#xa0;al., 2018</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">MFCC, Bagged Tree, KNN</td>
<td valign="middle" align="left">11 insects from 6 species</td>
<td valign="middle" align="left">species classification (over 97.1%); insect classification (over 92.3%)</td>
<td valign="middle" align="left">Limited sample size; Short-term features only</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B37">Phung et&#xa0;al., 2017</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">MFCC and LFCC, SVM</td>
<td valign="middle" align="left">InsectSingers</td>
<td valign="middle" align="left">Accuracy (99.08%)</td>
<td valign="middle" align="left">Limited to <italic>cicada</italic> species</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B34">Noda et&#xa0;al., 2016</xref>)</td>
</tr>
<tr>
<td valign="middle" align="left">MFCC, Probabilistic Neural Network</td>
<td valign="middle" align="left">insect sound library from agricultural research service of United States department of agriculture</td>
<td valign="middle" align="left">Accuracy (96%)</td>
<td valign="middle" align="left">Limited to 6 species</td>
<td valign="middle" align="left">(<xref ref-type="bibr" rid="B25">Le-Qing, 2011</xref>)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The methodologies employed in insect acoustic recognition and detection can be broadly categorized into traditional machine learning (ML) and deep learning (DL) techniques. Traditional ML methods, as exemplified by studies such as references (<xref ref-type="bibr" rid="B25">Le-Qing, 2011</xref>; <xref ref-type="bibr" rid="B34">Noda et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B37">Phung et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B35">Noda et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B4">Basak et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B8">de Souza et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B49">Wang and Vhaduri, 2024</xref>), rely on handcrafted features like mel frequency cepstral coefficients (MFCC) and linear frequency cepstral coefficients (LFCC), paired with classifiers such as support vector machines (SVM) and k-Nearest Neighbors (kNN). These approaches have achieved notable accuracy in constrained scenarios, for instance, <xref ref-type="bibr" rid="B37">Phung et&#xa0;al. (2017)</xref> reported 97.1% accuracy for 11 insect species, while <xref ref-type="bibr" rid="B34">Noda et&#xa0;al. (2016)</xref> achieved 99.08% accuracy for cicada detection. However, their reliance on manual feature engineering limits adaptability to complex or high-frequency acoustic patterns, as highlighted by the analysis (<xref ref-type="bibr" rid="B25">Le-Qing, 2011</xref>; <xref ref-type="bibr" rid="B37">Phung et&#xa0;al., 2017</xref>).</p>
<p>In contrast, DL techniques leverage automated feature learning and scalable architectures to overcome these limitations. Studies such as those in references (<xref ref-type="bibr" rid="B22">Karar et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B51">Zhang et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B45">Tey et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B5">Boulila et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B13">Fai&#xdf; and Stowell, 2023</xref>; <xref ref-type="bibr" rid="B43">Szekeres et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B46">Truong et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B50">Zhang, 2023</xref>; <xref ref-type="bibr" rid="B3">Balingbing et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B9">Dhanaraj et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B32">Montemayor et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B15">Ferreira et&#xa0;al., 2025</xref>) employ convolutional neural networks (CNNs), gated recurrent units (GRUs), transformers, and hybrid models, often integrated with IoT or unmanned aerial vehicle (UAV) systems. For instance, the deep multibranch fusion residual network (DMF-ResNet) (<xref ref-type="bibr" rid="B9">Dhanaraj et&#xa0;al., 2024</xref>), trained on the comprehensive &#x201c;Bug Bytes sound library&#x201d;, demonstrated exceptional performance with 99.75% accuracy, 99.18% precision, and 99.08% recall, showcasing the potential of DL for high-precision applications. Similarly, Kamar et&#xa0;al. (<xref ref-type="bibr" rid="B22">Karar et&#xa0;al., 2021</xref>) utilized the IoT-based Inception-Residual Network V2 (InceptionResNet-V2) model on the TreeVibes dataset, achieving 97.18% accuracy for real-time red palm weevil detection while addressing class imbalance through transfer learning. UAV-integrated systems (<xref ref-type="bibr" rid="B50">Zhang, 2023</xref>) further expanded scalability by combining 48MP visual data with acoustic recordings from 16 species to achieve 92% precision and 84% recall for large-area <italic>grasshopper</italic> monitoring. Adaptive frontends, such as the learnable frontend (LEAF) model (<xref ref-type="bibr" rid="B13">Fai&#xdf; and Stowell, 2023</xref>), dynamically adjusted filter parameters for high-frequency sounds, leading to an accuracy boost from 67% to 86% across multiple datasets, thereby outperforming static Mel-spectrogram method. Similarly, the CNN-Gated Recurrent Unit (GRU) model (<xref ref-type="bibr" rid="B46">Truong et&#xa0;al., 2023</xref>) enhanced bee buzzing recognition by 1% over existing methods, leveraging Bayesian optimization for hyperparameter tuning. Despite these advancements, DL models still face several challenges. One major issue is computational intensity, as evidenced by the IoT system (<xref ref-type="bibr" rid="B9">Dhanaraj et&#xa0;al., 2024</xref>), which requires significant resources. Another challenge is dataset dependency, highlighted by the transformer model (<xref ref-type="bibr" rid="B15">Ferreira et&#xa0;al., 2025</xref>), which necessitated data augmentation for 15 <italic>bee</italic> species. However, reliance on augmentation introduces risks, particularly the potential for overfitting to synthetic variations, as noted in Wang et&#xa0;al (<xref ref-type="bibr" rid="B49">Wang and Vhaduri, 2024</xref>). Additionally, environmental noise sensitivity remains a concern, as seen in the UAV recordings (<xref ref-type="bibr" rid="B50">Zhang, 2023</xref>), which suffered from interference.</p>
<p>The strengths and limitations of these methodologies are intricately shaped by the inherent characteristics of the datasets employed. Some studies focus on small, specialized datasets, such as the 43 acoustic recordings of three <italic>cicada</italic> species (<xref ref-type="bibr" rid="B45">Tey et&#xa0;al., 2022</xref>), datasets focused on <italic>fruit flies</italic> and <italic>mosquitoes</italic> (<xref ref-type="bibr" rid="B43">Szekeres et&#xa0;al., 2023</xref>), <italic>bee</italic> buzzing sounds (<xref ref-type="bibr" rid="B36">Phan et&#xa0;al., 2023</xref>), and 343 samples spanning three insects (<xref ref-type="bibr" rid="B35">Noda et&#xa0;al., 2019</xref>). While these datasets are useful for specific applications, they often lack the diversity required for robust generalization. Larger datasets, such as the TreeVibes database (<xref ref-type="bibr" rid="B22">Karar et&#xa0;al., 2021</xref>) and the InsectSet66 dataset (<xref ref-type="bibr" rid="B13">Fai&#xdf; and Stowell, 2023</xref>), offer a more comprehensive coverage of insect sounds. However, even these datasets face challenges, such as class imbalance and limited audio quality (<xref ref-type="bibr" rid="B22">Karar et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B13">Fai&#xdf; and Stowell, 2023</xref>). On the other hand, several studies highlighted limitations in scalability and dependency on specialized equipment. For instance, the dual-frequency and spectral fusion module (DFSM) architecture with EfficientNet (<xref ref-type="bibr" rid="B19">He et&#xa0;al., 2024</xref>) achieved 80.26% accuracy on InsectSet32, outperforming state-of-the-art (SOTA) methods by 3%, but scalability for large-scale monitoring remained an unresolved issue. Similarly, Zhang et&#xa0;al. (<xref ref-type="bibr" rid="B32">Montemayor et&#xa0;al., 2024</xref>) achieved high precision and recall using MFCC features and CNNs on 2,800 <italic>rice weevil</italic> samples but required high-performance microphones, limiting practical deployment. Innovations like micro-electromechanical system (MEMS) microphones (<xref ref-type="bibr" rid="B3">Balingbing et&#xa0;al., 2024</xref>) enabled non-chemical pest detection with 84.51% accuracy but faced challenges with noisy storage environments and limited taxonomic coverage.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Data description</title>
<p>The comparison of various insect sound datasets is presented in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, highlighting the species covered, sample sizes, descriptions, and download links. These datasets differ in terms of the number of species, total samples, and species diversity. In comparison, datasets such as BUZZ1 (<xref ref-type="bibr" rid="B24">Kulyukin et&#xa0;al., 2018</xref>), BUZZ2 (<xref ref-type="bibr" rid="B24">Kulyukin et&#xa0;al., 2018</xref>), SINA (<xref ref-type="bibr" rid="B48">Walker and Moore, 2019</xref>), and Insectsingers (<xref ref-type="bibr" rid="B31">Marshall and Hill</xref>), focus on a limited number of insect species, resulting in less diversity in terms of species variety and sound level range. Additionally, datasets like ESC-50 (<xref ref-type="bibr" rid="B38">Piczak, 2015</xref>), and InsectSet32 (<xref ref-type="bibr" rid="B12">Fai&#xdf;, 2022</xref>), focus on fewer species with smaller sample sizes. The Bug Bytes sound library (<xref ref-type="bibr" rid="B29">Mankin, 2019</xref>), features a broader range of insect species (72), with 7,200 samples. However, the presence of non-agricultural pest species may introduce interference, limiting the dataset&#x2019;s effectiveness for agricultural pest monitoring.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison of various insect sound datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">References</th>
<th valign="middle" align="left">Insect species</th>
<th valign="middle" align="left">Number of samples</th>
<th valign="middle" align="left">Brief description</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">BUZZ1 (<xref ref-type="bibr" rid="B24">Kulyukin et&#xa0;al., 2018</xref>)<break/>Available online:<break/>
<ext-link ext-link-type="uri" xlink:href="https://usu.app.box.com/v/BeePiAudioData">https://usu.app.box.com/v/BeePiAudioData</ext-link> (accessed on 11 May 2021).</td>
<td valign="middle" align="left">
<italic>Bee\Cricket\Noise</italic>
</td>
<td valign="middle" align="left">3300\3500\3460</td>
<td valign="middle" align="left">Very few insect species</td>
</tr>
<tr>
<td valign="middle" align="left">BUZZ2 (<xref ref-type="bibr" rid="B24">Kulyukin et&#xa0;al., 2018</xref>)<break/>Available online:<break/>
<ext-link ext-link-type="uri" xlink:href="https://usu.app.box.com/v/BeePiAudioData">https://usu.app.box.com/v/BeePiAudioData</ext-link> (accessed on 11 May 2021).</td>
<td valign="middle" align="left">
<italic>Bee\Cricket\Noise</italic>
</td>
<td valign="middle" align="left">4300\4500\4114</td>
<td valign="middle" align="left">Very few insect species</td>
</tr>
<tr>
<td valign="middle" align="left">SINA (<xref ref-type="bibr" rid="B48">Walker and Moore, 2019</xref>)<break/>Available online:<break/>
<ext-link ext-link-type="uri" xlink:href="https://orthsoc.org/sina/crickets.htm">https://orthsoc.org/sina/crickets.htm</ext-link>.<break/>(accessed on 7 September 2025)</td>
<td valign="middle" align="left">255 species of <italic>katydids, crickets</italic> and <italic>cicadas</italic>
</td>
<td valign="middle" align="left">/</td>
<td valign="middle" align="left">Only three insect species</td>
</tr>
<tr>
<td valign="middle" align="left">Insectsingers (<xref ref-type="bibr" rid="B31">Marshall and Hill</xref>)<break/>Available online:<break/>
<ext-link ext-link-type="uri" xlink:href="https://www.insectsingers.com/">https://www.insectsingers.com/</ext-link>.<break/>(accessed on 7 September 2025)</td>
<td valign="middle" align="left">343 species of <italic>katydids, crickets</italic> and <italic>cicadas</italic>
</td>
<td valign="middle" align="left">/</td>
<td valign="middle" align="left">Only three insect species</td>
</tr>
<tr>
<td valign="middle" align="left">ESC-50 (<xref ref-type="bibr" rid="B38">Piczak, 2015</xref>)<break/>Available online:<break/>
<ext-link ext-link-type="uri" xlink:href="https://github.com/karolpiczak/ESC-50?tab=readme-ov-file">https://github.com/karolpiczak/ESC-50?tab=readme-ov-file</ext-link>.<break/>(accessed on 7 September 2025)</td>
<td valign="middle" align="left">
<italic>Frog</italic>\Insects(flying)\<italic>Criets</italic>\<italic>Chirping birds</italic>
</td>
<td valign="middle" align="left">40 examples per class, each 5 seconds long</td>
<td valign="middle" align="left">No subdivision of insect species</td>
</tr>
<tr>
<td valign="middle" align="left">InsectSet32 (<xref ref-type="bibr" rid="B12">Fai&#xdf;, 2022</xref>)<break/>Available online:<break/>
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.5281/zenodo.7072196">https://doi.org/10.5281/zenodo.7072196</ext-link>.<break/>(accessed on 7 September 2025)</td>
<td valign="middle" align="left">9 species of <italic>Orthoptera</italic> and 23 species of <italic>Cicadidae</italic>
</td>
<td valign="middle" align="left">335 files, 57 minutes in total</td>
<td valign="middle" align="left">Only two insect species</td>
</tr>
<tr>
<td valign="middle" align="left">Bug Bytes sound library (<xref ref-type="bibr" rid="B29">Mankin, 2019</xref>)<break/>Available online:<break/>
<ext-link ext-link-type="uri" xlink:href="https://data.nal.usda.gov/dataset/bug-bytes-sound-library-stored-product-insect-pest-sounds">https://data.nal.usda.gov/dataset/bug-bytes-sound-library-stored-product-insect-pest-sounds</ext-link>. (accessed on 7 September 2025)</td>
<td valign="middle" align="left">72</td>
<td valign="middle" align="left">7200</td>
<td valign="middle" align="left">Includes many non-agricultural pest insects, with diverse categories that may cause interference</td>
</tr>
<tr>
<td valign="middle" align="left">InsectSound1000 (<xref ref-type="bibr" rid="B6">Branding et&#xa0;al., 2024</xref>)<break/>Available online:<break/>
<ext-link ext-link-type="uri" xlink:href="https://www.openagrar.de/receive/openagrar_mods_00091171">https://www.openagrar.de/receive/openagrar_mods_00091171</ext-link>.<break/>(accessed on 7 September 2025)</td>
<td valign="middle" align="left">
<italic>Aphidoletes aphidimyza, Myzus persicae</italic>, et&#xa0;al., a total of 12 species.</td>
<td valign="middle" align="left">Over169,000 labelled samples</td>
<td valign="middle" align="left">Diverse insect species with a wide range of sound levels.</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Notably, InsectSound1000 (<xref ref-type="bibr" rid="B6">Branding et&#xa0;al., 2024</xref>) stands out as the most comprehensive and diverse dataset for training robust insect acoustic recognition models, especially given its extensive sample size and wide range of insect species and sound levels. Therefore, InsectSound1000 is selected for this study, containing over 169,000 labelled sound samples from 12 insect species, recorded in an anechoic box with a four-channel low-noise microphone array. The acoustic intensity spans a range from the very loud <italic>Bombus terrestris</italic> to the nearly inaudible <italic>Aphidoletes aphidimyza</italic> for the human ears. Each sample is a four-channel WAV file with a duration of 2500 ms, sampled at 16 kHz with 32-bit resolution. With over 1000 hours of high-quality recordings, InsectSound1000 is suitable for training DL models for insect acoustic recognition. Primarily used for model pre-training, this dataset also supports developing insect acoustic recognition systems across different hardware platforms for various species. As outlined in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, the details of the pests used in the analytical procedures are provided. To ensure data balance and consistency, the final dataset is determined based on the class with the smallest sample size.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Details of Pest used in analytical procedures.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Insect order</th>
<th valign="middle" align="left">Insect family</th>
<th valign="middle" align="left">Insect species</th>
<th valign="middle" align="left">Number</th>
<th valign="middle" align="left">Duration /ms</th>
<th valign="middle" align="left">Brief description</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">
<italic>Diptera</italic>
</td>
<td valign="middle" align="left">
<italic>Cecidomyiidae</italic>
</td>
<td valign="middle" align="left">
<italic>Aphidoletes aphidimyza</italic>
</td>
<td valign="middle" align="left">14065</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A natural predator of <italic>aphids</italic>, whose sound can be used for pest control monitoring.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Hymenoptera</italic>
</td>
<td valign="middle" align="left">
<italic>Apidae</italic>
</td>
<td valign="middle" align="left">
<italic>Bombus terrestris</italic>
</td>
<td valign="middle" align="left">18291</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A <italic>bumblebee</italic> species, whose buzzing sounds can be used to monitor pollinator activity and assess crop health.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Diptera</italic>
</td>
<td valign="middle" align="left">
<italic>Sciaridae</italic>
</td>
<td valign="middle" align="left">
<italic>Bradysia difformis</italic>
</td>
<td valign="middle" align="left">11394</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A small <italic>fungus gnat</italic>, whose larvae feed on <italic>fungi</italic> and damage the root systems of host plants in humid environments.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Coleoptera</italic>
</td>
<td valign="middle" align="left">
<italic>Coccinellidae</italic>
</td>
<td valign="middle" align="left">
<italic>Coccinella septempunctata</italic>
</td>
<td valign="middle" align="left">14682</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A beneficial <italic>ladybug</italic> that feeds on <italic>aphids</italic> and helps control agricultural pests.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Diptera</italic>
</td>
<td valign="middle" align="left">
<italic>Syrphidae</italic>
</td>
<td valign="middle" align="left">
<italic>Episyrphus balteatus</italic>
</td>
<td valign="middle" align="left">16868</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A <italic>hoverfly</italic> species, whose larvae feed on <italic>aphids</italic>, helping control agricultural pests.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Heteroptera</italic>
</td>
<td valign="middle" align="left">
<italic>Pentatomidae</italic>
</td>
<td valign="middle" align="left">
<italic>Halyomorpha halys</italic>
</td>
<td valign="middle" align="left">19671</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A destructive <italic>stink bug</italic>, whose feeding on plant tissues causes damage to crops, leading to yield loss and quality degradation.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Hemiptera</italic>
</td>
<td valign="middle" align="left">
<italic>Aphididae</italic>
</td>
<td valign="middle" align="left">
<italic>Myzus persicae</italic>
</td>
<td valign="middle" align="left">3208</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A destructive <italic>aphid</italic> that feeds on various crops, causing poor growth, yellowing leaves, and spreading plant viruses.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Heteroptera</italic>
</td>
<td valign="middle" align="left">
<italic>Pentatomidae</italic>
</td>
<td valign="middle" align="left">
<italic>Nezara viridula</italic>
</td>
<td valign="middle" align="left">20323</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A widespread agricultural pest that feeds on crops, causing damage to plant tissues and reduced yield.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Heteroptera</italic>
</td>
<td valign="middle" align="left">
<italic>Pentatomidae</italic>
</td>
<td valign="middle" align="left">
<italic>Palomena prasina</italic>
</td>
<td valign="middle" align="left">27340</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A <italic>shield bug</italic> that feeds on plant tissues, causing minor damage to crops and deformed fruits.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Heteroptera</italic>
</td>
<td valign="middle" align="left">
<italic>Pentatomidae</italic>
</td>
<td valign="middle" align="left">
<italic>Rhaphigaster nebulos</italic>
</td>
<td valign="middle" align="left">13443</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A <italic>shield bug</italic> that can cause minor damage to crops by feeding on plant tissues in certain circumstances.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Hemiptera</italic>
</td>
<td valign="middle" align="left">
<italic>Aleyrodidae</italic>
</td>
<td valign="middle" align="left">
<italic>Vaporariorum</italic>
</td>
<td valign="middle" align="left">1062</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A common greenhouse pest that damages plants by feeding on sap and can spread viruses, negatively affecting crop growth and yield.</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>Lepidoptera</italic>
</td>
<td valign="middle" align="left">
<italic>Gelechiidae</italic>
</td>
<td valign="middle" align="left">
<italic>Tuta absoluta</italic>
</td>
<td valign="middle" align="left">633</td>
<td valign="middle" align="left">2500</td>
<td valign="middle" align="left">A destructive pest of <italic>tomato</italic> crops, whose larvae damage leaves and fruits, leading to significant yield and quality losses.</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4">
<label>4</label>
<title>Proposed methodology</title>
<p>The objective of this study is to rapidly identify and detect pests at their early stages of appearance or as soon as they reach detectable levels, enabling the swift initiation of localized control measures to prevent further spread and minimize damage. For instance, upon detecting a small number of pests in a specific area of an orchard, the affected zone is immediately isolated. Subsequently, biological control methods or precision pesticide treatments are applied to eradicate the pests at this early stage, thereby preventing their spread to the entire orchard. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> describes the architecture of the proposed approach. The process begins with pre-processing of the InsectSound1000 dataset, followed by PLMS feature extraction alongside conventional feature representations for comparative analysis. These feature representations are mapped into DL-based classification frameworks, where our proposed model is systematically benchmarked against ResNet18, EfficientNet, VGG19, DenseNet, and MobileNet and other algorithms to assess its efficacy and robustness. The resulting classification facilitates high-precision pest identification, enhancing automated surveillance and enabling proactive early-stage intervention strategies.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Architecture of the proposed approach for insect acoustic classification with comparative analysis of multiple features and benchmark models.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g001.tif">
<alt-text content-type="machine-generated">Diagram outlining the process of classifying insect sounds. It starts with the InsectSound1000 database, followed by pre-processing steps like low-pass filtering and downsampling. Features are extracted using methods such as wavelet transform, STFT, and PLMS. These features are input into models like Resnet18, EfficientNet, Densenet, VGG19, and Mobilenet for classification, identifying target pests.</alt-text>
</graphic>
</fig>
<sec id="s4_1">
<label>4.1</label>
<title>Preprocessing</title>
<sec id="s4_1_1">
<label>4.1.1</label>
<title>Low-pass filtering</title>
<p>The purpose of low-pass filtering is to eliminate components of the signal above specific frequency. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> presents the representative spectrums of the insect acoustic signals from <italic>Aphidoletes aphidimyza</italic> and <italic>Bradysia difformis</italic>. Analysis shows that the acoustic signals of the pests studied in this paper are predominantly concentrated within the low-frequency range.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Spectra of the insect acoustic signals from <italic>Aphidoletes aphidimyza</italic> and <italic>Bradysia difformis</italic>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g002.tif">
<alt-text content-type="machine-generated">Two side-by-side graphs show sound pressure level against frequency. The left graph features a more even distribution with gradual decreases, while the right graph has distinct peaks and quieter noise levels. Both graphs range from zero to eight thousand Hertz on the x-axis and differing decibel levels on the y-axis.</alt-text>
</graphic>
</fig>
<p>Therefore, finite impulse response (FIR) filter is employed for low-pass filtering to suppress high-frequency noise while effectively preserving the essential low-frequency information. FIR filters, a class of digital filters with finite-duration impulse responses, are distinguished by their ideal linear phase characteristics, where the output is computed as the convolution of the input signal with the filter coefficients, as depicted in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The discrete-time output of an FIR filter can be represented as the convolution of the input signal and the filter coefficients. Where <italic>N</italic> denotes the order of the filter, i.e. the number of filter coefficients. The design of an FIR filter typically involves three main steps: determining the filter type, specifying the cutoff frequency, and selecting an appropriate window function. In practical implementations, the coefficients of the FIR filter are commonly calculated using the window function method. This approach applies a window function to the impulse response of the ideal filter, effectively suppressing sidelobe leakage and ripple effects, thereby improving overall filter performance. In this study, low-pass filtering is selected. The time domain impulse response of an ideal low-pass filter is calculated as shown in <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>ideal</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>sin</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the cutoff frequency. The ideal filter possesses an impulse response of infinite duration, rendering it infeasible for practical implementation. Therefore, it is necessary to truncate its duration through the window function method. The Hamming window can effectively reduce the sidelobe leakage and enhance the stopband attenuation performance. The application of the Hamming window yields a flatter frequency response within the passband, facilitates more rapid attenuation within the stopband, and effectively suppresses sidelobe levels. The Hamming window is calculated as shown in <xref ref-type="disp-formula" rid="eq3">Equation 3</xref>:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>0.54-0</mml:mn>
<mml:mtext>.46cos(</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>)</mml:mo>
<mml:mo>,</mml:mo>
<mml:mn>&#xa0;0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The ideal impulse response is multiplied by the window function to obtain the final filter coefficients, as shown in <xref ref-type="disp-formula" rid="eq4">Equation 4</xref>:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>ideal</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In addition, the choice of filter order is influenced by the transition band width, passband ripple, stopband attenuation, and other design parameters. A higher order results in a narrower transition band and more accurate frequency response, but it also increases computational complexity.</p>
</sec>
<sec id="s4_1_2">
<label>4.1.2</label>
<title>Downsampling</title>
<p>In accordance with the Nyquist sampling theorem, the maximum frequency component of the signal must be less than half of the target sampling rate. If high-frequency components are not adequately attenuated prior to downsampling, aliasing artifacts may arise, wherein high-frequency energy is folded into the lower frequency spectrum, leading to significant distortion of the signal. Consequently, applying low-pass filtering before downsampling is essential to prevent aliasing effects.</p>
<p>Let <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the maximum frequency of the signal. The target sampling rate <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> must satisfy the following condition, as shown in <xref ref-type="disp-formula" rid="eq5">Equation 5</xref>:</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Downsampling, an essential method in signal processing, is employed to reduce data dimensionality and mitigate computational load. Assume that the downsampling function is defined as <italic>Resample</italic>, which is defined as <xref ref-type="disp-formula" rid="eq6">Equation 6</xref>:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mtext>y</mml:mtext>
<mml:mo>'</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>Re</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the sampling rate of the original signal <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the sampling rate of the target signal <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:mtext>y</mml:mtext>
<mml:mo>'</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. <italic>R</italic> denotes the sampling rate ratio. Since <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is usually not an integer, interpolation must be performed. This paper utilizes the <italic>sinc</italic> interpolation method, a high-fidelity anti-aliasing resampling technique that integrates a window function to effectively suppress sidelobe leakage. The formula is defined as shown in <xref ref-type="disp-formula" rid="eq7">Equation 7</xref>:</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mtext>y</mml:mtext>
<mml:mo>'</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>k</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mfrac>
<mml:mi>n</mml:mi>
<mml:mi>R</mml:mi>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mstyle>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;&#xa0;</mml:mtext>
<mml:mi>sin</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>sin</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the window function. The primary objective of downsampling is to reduce the number of sampling points, thereby lowering the computational burden associated with subsequent feature extraction and classifier model training. However, higher downsampling ratios may compromise temporal resolution, particularly affecting the accurate representation of high-frequency components. Therefore, the choice of an appropriate downsampling rate must achieve a trade-off between preserving the critical spectral characteristics of the signal and satisfying the computational efficiency requirements of downstream processing tasks.</p>
</sec>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Feature engineering</title>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>PLMS</title>
<p>An innovative feature representation method, termed the PLMS, is proposed in this study to
enhance the accuracy of insect sound signal recognition. The complete process of PLMS extraction is
outlined in <xref ref-type="boxed-text" rid="algo1">
<bold>Algorithm 1</bold>
</xref>. First, the original insect acoustic signals are preprocessed using low-pass filtering and downsampling to reduce computational complexity while preserving relevant spectral content. The processed signals are then segmented into overlapping patch-level windows with predefined window length and shift, enabling the extraction of local temporal structural features. Since these signals exhibit distinct spectral characteristics across different temporal regions, dividing the spectrogram into smaller, localized patches allows for the capture of subtle, context-specific variations in both time and frequency domains. This localized analysis enhances the model&#x2019;s ability to detect fine-grained temporal dynamics, which is crucial for improving the accuracy and robustness of classification and recognition tasks. Furthermore, by enabling the model to learn discriminative features from multiple local regions, the patch-level approach contributes to better generalization performance when applied to unseen data. The specific formula for the patch-level segmentation operation is as shown in <xref ref-type="disp-formula" rid="eq9">Equation 8</xref>:</p>
<boxed-text id="algo1" position="float">
<label>Algorithm 1</label>
<title>PLMS extraction.</title>
<p><graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g010.tif"/></p>
</boxed-text>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mtext>y</mml:mtext>
<mml:msub>
<mml:mo>'</mml:mo>
<mml:mrow>
<mml:mtext>frame</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>y</mml:mtext>
<mml:mo>'</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>'</mml:mo>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>N</italic> and <italic>shift</italic> represent the patch-level window length and shift, respectively. Subsequently, the short-time fourier transform (STFT) is performed on each patch-level segmented signal to extract localized time-frequency features. This process facilitates the characterization of spectral variations within each temporal segment, thereby enhancing the representation of non-stationary signal components, as shown in <xref ref-type="disp-formula" rid="eq9">Equation 9</xref>:</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:mrow>
<mml:mtext>y</mml:mtext>
<mml:msub>
<mml:mo>'</mml:mo>
<mml:mrow>
<mml:mtext>frame</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>n</italic> denotes the frame index. &#x1d458; denotes the frequency index. <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the STFT complex spectrum of the &#x1d45b;-th frame.</p>
<p>Next, in order to compute the Mel spectrum for each frame, a bank of Mel-scale filters must first be constructed to project the power spectrum onto the perceptually motivated Mel frequency. The Mel filter bank comprises a series of triangular band-pass filters that are uniformly spaced on the Mel frequency scale but nonlinearly distributed along the linear frequency axis. Given <italic>M</italic> Mel filters, the frequency response <italic>m</italic>-th filter is defined as shown in <xref ref-type="disp-formula" rid="eq10">Equation 10</xref>:</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mtext>m</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&lt;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&lt;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&lt;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mtext>&#xa0;&#xa0;&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the k-th frequency point. <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the center frequency of the <italic>m</italic>-th Mel filter. <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the boundary frequencies of adjacent filters. Center frequency is usually mapped from the Mel scale to the frequency axis using the following formula as shown in <xref ref-type="disp-formula" rid="eq11">Equation 11</xref>:</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mtext>el</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mfrac>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mtext>el</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mtext>el</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mtext>el</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where the Mel transform is defined as shown in <xref ref-type="disp-formula" rid="eq12">Equation 12</xref>:</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mtext>el</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>2595</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>log</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>700</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Then Mel filter bank and power spectrum are weighted superimposed frame by frame to obtain Mel spectrum, as shown in <xref ref-type="disp-formula" rid="eq13">Equation 13</xref>:</p>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:munderover>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the power spectrum of each frame, computed as follows shown in <xref ref-type="disp-formula" rid="eq14">Equation 14</xref>:</p>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>|</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:msup>
<mml:mo>|</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The Mel frequency scale, a nonlinear transformation of frequency, provides enhanced resolution during the low-frequency range while compressing resolution during the high-frequency domain. This property aligns well with the spectral characteristics of insect acoustic signals, where the energy is predominantly concentrated in the low-frequency components. To compress the dynamic range of the Mel spectrogram, the logarithmic transformation and normalization are applied, mitigating the influence of high-amplitude frequency components and highlighting finer details during low-energy regions, as shown in <xref ref-type="disp-formula" rid="eq15">Equation 15</xref>:</p>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mi>log</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>10</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>log</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;ref</mml:mtext>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>S</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Finally, to align the visual representation of the Mel spectrogram with the logarithmic nature of human auditory perception, a logarithmic transformation is applied to its frequency axis. This scaling enhances the interpretability of spectral content, particularly in lower frequency regions where human sensitivity is greater. By mapping the linear frequency axis to a logarithmic scale, the resulting spectrogram more accurately reflects perceptual frequency resolution. Furthermore, this approach not only clarifies the low-frequency regions but also diminishes the visual dominance of high-frequency details, leading to a more balanced signal representation. Assume <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>log</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the transformed logarithmic frequency, the transformation formula is shown as <xref ref-type="disp-formula" rid="eq16">Equation 16</xref>:</p>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>log</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>log</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>From <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, the original audio waveform (<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;3A</bold>
</xref>) and the PLMS spectrogram (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>) are presented for comparison. While the original waveform offers a broad, global representation of signal energy variations over time, it falls short in capturing the finer, localized details and intricate frequency characteristics inherent within the signal. In contrast, the PLMS provides a more refined and precise depiction by explicitly encoding time-frequency patterns through a hierarchical decomposition, thereby revealing essential spectral and temporal features that the waveform alone cannot fully convey.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Comparison of original acoustic waveform and PLMS spectrogram. <bold>(A)</bold> Original audio waveform. <bold>(B)</bold> PLMS spectrogram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g003.tif">
<alt-text content-type="machine-generated">Panel A is a waveform graph showing amplitude over time, indicating sound fluctuations. Panel B is a spectrogram depicting frequency in Hertz over time in seconds, with a color scale indicating decibel levels from zero to negative eighty.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> illustrates a comparative analysis of PLMS spectrogram of <italic>Bombus terrestris</italic> acoustic signals under varying sampling rates and patch size parameters. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4A</bold>
</xref> presents the PLMS with a 16,000 Hz sampling rate and patch size of 10, covering a wide frequency range and preserving high-frequency components. However, the insect acoustic signals analyzed exhibit spectral energy predominantly concentrated in the low-frequency region, making high-frequency contributions relatively insignificant. Moreover, the high sampling rate imposes constraints on temporal resolution, leading to less detailed local feature representation in the high-frequency region. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4B</bold>
</xref> shows the PLMS with a reduced sampling rate of 2,500 Hz while maintaining a patch size of 10. This configuration decreases frequency resolution and slightly reduces the clarity of low-frequency details but significantly enhances temporal resolution, which enables more precise characterization of transient spectral dynamics and is critical for capturing short-duration acoustic events in insect signals. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4C</bold>
</xref> retains the 2,500 Hz sampling rate while increasing the patch size to 20. This adjustment markedly improves frequency resolution, producing smoother and more continuous spectral structures across the full frequency spectrum. However, the improved frequency resolution comes at the expense of temporal resolution, diminishing the ability to resolve rapid temporal variations. In summary, higher sampling rates facilitate the preservation of high-frequency spectral information but require a trade-off with temporal resolution. Larger patch sizes improve frequency resolution at the cost of temporal precision. Thus, the selection of sampling rate and patch size should be task-specific, balancing time and frequency resolution to achieve optimal and accurate feature representation.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>PLMS spectrograms of <italic>Bombus terrestris</italic> across different hyperparameters. <bold>(A)</bold> 16000Hz &amp; patch size 10 <bold>(B)</bold> 2500Hz &amp; patch size 10 <bold>(C)</bold> 2500Hz &amp; patch size 20.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g004.tif">
<alt-text content-type="machine-generated">Three spectrograms labeled A, B, and C are displayed. A highlights a high frequency region with a black arrow pointing to higher frequencies. B shows low frequency resolution with a black rectangle and a high time resolution area indicated by a white rectangle. C depicts high frequency resolution marked by a black arrow and low time resolution noted with a white arrow and rectangle. Each spectrogram includes a color scale indicating decibel levels from -80 dB to 0 dB.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> presents a comparison of the acoustic signal spectrograms from <italic>Bombus terrestris</italic> and <italic>Bradysia difformis</italic>, before and after logarithmic scaling. As shown in <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5A, B</bold>
</xref>, the logarithmic transformation significantly amplifies and highlights the energy in the low-frequency region, effectively expanding the dynamic range in this frequency band and thereby rendering its details more discernible. In contrast, the spectrograms without logarithmic scaling in <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5C, D</bold>
</xref> exhibit relatively flat low-frequency energy with a limited dynamic range, which results in the masking of low-frequency details and hinders the effective capture of subtle frequency variations. The PLMS feature, by applying logarithmic scaling, effectively compresses the influence of high-amplitude frequency components while enhancing the resolution of low-energy regions. This leads to improved robustness and representational capacity of the features, facilitating subsequent acoustic feature extraction and classification. Therefore, the logarithmic transformation of PLMS constitutes a crucial preprocessing step for enhancing the analysis of insect acoustic signals, markedly improving the expressiveness and discriminability of low-frequency components.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Comparison of spectrograms before and after logarithmic scaling for <italic>Bombus terrestris</italic> and <italic>Bradysia difformis.</italic> <bold>(A)</bold> <italic>Bombus terrestris</italic> with logarithmic scaling <bold>(B)</bold> <italic>Bradysia difformis</italic> with logarithmic scaling. <bold>(C)</bold> <italic>Bombus terrestris</italic> without logarithmic scaling <bold>(D)</bold> <italic>Bradysia difformis</italic> without logarithmic scaling.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g005.tif">
<alt-text content-type="machine-generated">Spectrograms labeled A to D display frequency against time with color gradients indicating decibel levels from 0 to -80. A and B highlight low-frequency amplification regions; C and D point to low-frequency regions.</alt-text>
</graphic>
</fig>
<p>To summarize, the PLMS representation achieves an optimal balance between computational efficiency and feature expressiveness by integrating downsampling with logarithmic scaling. While downsampling effectively reduces computational burden, logarithmic scaling enhances the fidelity of low-frequency components that are essential for capturing nuanced bioacoustic features. Additionally, patch-level processing partitions the spectrogram into localized sub-regions, thereby amplifying transient and harmonic characteristics intrinsic to insect acoustic signals. By maintaining the inherent time-frequency continuity within these localized spectro-temporal segments, PLMS markedly improves the model&#x2019;s capability to characterize the dynamic and non-stationary properties of bioacoustic signals. Collectively, these methodological innovations produce a robust and discriminative feature representation that significantly elevates classification performance across a diverse spectrum of complex bioacoustic applications.</p>
</sec>
<sec id="s4_2_2">
<label>4.2.2</label>
<title>Baseline features</title>
<p>In this study, we select five features as baseline features and juxtaposed them with the PLMS features. These five features encompass the STFT, Wavelet Transform, Wigner-Ville Distribution, Generalized S-Transform and LEAF. The extraction procedures for these features are depicted in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Flowchart of the feature extraction procedures for different acoustic representations.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g006.tif">
<alt-text content-type="machine-generated">Flowchart illustrating different audio signal processing techniques. It begins with &#x201c;Input audio signal&#x201d; split into five paths: STFT, Wavelet Transform, Wigner-Ville, Generalized S Transform, and LEAF. Each path involves specific steps like framing, windowing, convolution, and transform computations, leading to distributions or spectrogram generation.</alt-text>
</graphic>
</fig>
<p>The STFT partitions the acoustic signal into consecutive short-time windows and applies the Fourier Transform within each window, generating the time-frequency representation. One of the key advantages of the STFT is its capability to offer localized time-frequency representations. However, the Heisenberg uncertainty principle introduces an inherent trade-off between time and frequency resolution, limiting the precision with which both can be simultaneously captured. The Wavelet Transform, in contrast, decomposes the signal using wavelet basis functions across multiple scales, enabling excellent localization in both the time and frequency domains. The ability of the Wavelet Transform to capture transient and localized variations within the signal makes it particularly effective, especially for tasks where traditional Fourier analysis fails due to its inability to resolve short-lived or time-varying features. The Wigner-Ville Distribution, a joint time-frequency analysis technique, offers high-resolution time-frequency maps, addressing the typical limitations of classical linear time-frequency methods in reconciling time and frequency resolution. However, the presence of cross-term interference can severely degrade the clarity of the time-frequency map, impairing the accuracy of the analysis. In contrast, the Generalized S-Transform provides adaptive localization by selecting an appropriate kernel function, offering greater flexibility in adjusting time and frequency resolution. Such capabilities make the method particularly well-suited for analyzing non-stationary signals with rapid and substantial variations in instantaneous frequency, which pose challenges for traditional methods that may fail to capture these dynamic characteristics. LEAF employs a learnable convolutional frontend to optimize time-frequency representations through end-to-end training. By replacing fixed filterbanks and compressors with trainable Gabor filters and adaptive per-channel energy normalization, it enables task-specific feature extraction and robust noise suppression.</p>
</sec>
<sec id="s4_2_3">
<label>4.2.3</label>
<title>Cross-modal transfer learning with pretrained YOLO</title>
<p>The YOLO series of models, known for their single-stage detection architecture, have been widely adopted in computer vision due to their high-speed and accurate object detection capabilities. As a more recent iteration in this series, YOLOv11 retains the core real-time detection advantages while introducing enhanced feature extraction structures, stronger attention mechanisms, and a more lightweight design. Given the computational constraints and deployment requirements of pest acoustic classification tasks, this study adopts the YOLOv11n variant. This version maintains high accuracy while significantly reducing model parameters and computational overhead, making it well-suited for efficient inference on edge devices. Therefore, leveraging the preprocessed PLMS feature spectrograms, we employ the pre-trained YOLOv11n-cls model within a transfer learning framework and perform partial fine-tuning, updating the classification head and higher-level feature layers. This targeted adaptation preserves the generic low-level features learned during pre-training while fine-tuning the higher-level representations for the specific insect acoustics classification task, yielding an end-to-end, high-efficiency classification framework.</p>
<p>As depicted in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>, we first optimize the input layer. The PLMS time-frequency representation of insect sounds, with dimensions 256&#xd7;256&#xd7;3, is preprocessed to comply with the input requirements of YOLOv11n, facilitating direct processing of multi-scale time-frequency features. Specifically, we apply resizing and normalization techniques, along with various data augmentation methods, including random cropping, random rotation, and color jittering. These strategies augment the diversity and robustness of the training dataset, thereby strengthening the model&#x2019;s generalization ability and significantly enhancing its accuracy in capturing the intrinsic variability of insect sound patterns.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Architecture of the cross-modal transfer learning model with pretrained YOLO.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g007.tif">
<alt-text content-type="machine-generated">Flowchart depicting a deep learning model for pest recognition from audio data. The process involves pre-processing with image loading, size adjustment, normalization, and data augmentation. This is followed by the backbone with convolutional layers and C3k2, C2PSA modules. The classify stage includes pooling, dropout, linear, and softmax layers. Outputs include multiple pest types identified by audio patterns.</alt-text>
</graphic>
</fig>
<p>Next, the augmented data is fed into the Backbone layer of the pre-trained model. We refine the Backbone architecture by introducing the C3 module with kernel size 2 (C3k2) structure, replacing several conventional convolutional layers. The C3k2 module represents a deep optimization of the traditional Cross Stage Partial Network (CSP) Bottleneck structure, aiming to improve feature extraction efficiency through parallel convolution designs and flexible parameter configurations. This module divides the input feature map into two branches: one is passed directly through to preserve shallow details, while the other undergoes multi-scale feature extraction via the C3k module, which employs variable convolution kernels, such as 3&#xd7;3 or 5&#xd7;5. The extracted features are then concatenated and fused. This design not only captures subtle high-frequency vibrations but also suppresses low-frequency environmental noise, ensuring a robust representation of acoustic features. Furthermore, this approach reduces redundant computations, accelerates inference speed, and employs grouped convolutions and channel compression for lightweight optimization. These enhancements make the model particularly well-suited for deployment on IoT edge devices in agricultural fields, offering high precision with minimal resource consumption. Additionally, we integrate the Cross Stage Partial with Pyramid Squeeze Attention (C2PSA) module into the Backbone. Based on the CSP structure, this module segments feature processing and incorporates the Pyramid Slice Attention mechanism to dynamically adjust spatial attention. Through multi-scale convolution kernels and channel weighting, the module significantly enhances the expression of the PLMS time-frequency dynamics, thereby improving the model&#x2019;s sensitivity to the frequency patterns of pest activities.</p>
<p>Finally, during the design of the classification module, we adopt a hybrid approach that combines global feature extraction with classification output, while freezing the detection head parameters to facilitate the expansion of classification-derived tasks. This module comprises convolutional layers, pooling layers, Dropout layers, linear layers, and Softmax layers: the convolutional layers extract high-dimensional features, the pooling layers perform downsampling to reduce dimensionality, the Dropout layers prevent overfitting, the linear layers map the features to the class space, and the Softmax layer outputs the class probabilities. This design strikes an optimal balance between computational efficiency and classification accuracy, making it well-suited for real-time classification tasks in complex environments. Consequently, the model proposed in this paper can effectively extract deep semantic information from PLMS, offering significant advantages over traditional methods in terms of parameter count, computational cost, and deployment efficiency.</p>
<p>The detailed parameter configuration of the proposed model is presented in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>. This model comprises 11 sequential stages, integrating five convolutional layers (Conv), three C3K2 modules, and one C2PSA module to enhance feature extraction efficiency while maintaining an effective trade-off between accuracy and computational complexity. The backbone begins with two Conv layers utilizing 3&#xd7;3 kernels with a stride of 2 to extract low-level features while reducing spatial dimensions. To enhance feature extraction, the C3K2 module is introduced in Stage 3, incorporating multi-scale convolutional kernels, though without residual connections, and applying a channel reduction ratio of 0.25 to optimize efficiency. Stage 4 follows with another Conv layer, further refining feature maps, while Stage 5 reintroduces the C3K2 module. As the model progresses, Stage 6 applies a Conv layer, followed by Stage 7, where a C3K2 module integrates residual connections to improve gradient flow and feature learning. Stage 8 continues with Conv processing, while Stage 9 employs another residual-connected C3K2 module to maintain deeper feature representations. The C2PSA module in Stage 10 enhances spatial attention through pyramid slice attention mechanisms, refining classification performance. Finally, Stage 11 serves as the classification layer, reducing feature maps to 12 output categories corresponding to different pest species. This architecture is designed to efficiently extract multi-scale features while maintaining robust classification performance for pest detection.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>The detailed parameter configuration of our model.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Stage</th>
<th valign="middle" align="center">Operator</th>
<th valign="middle" align="center">Filter</th>
<th valign="middle" align="center">Kernel</th>
<th valign="middle" align="center">Stride</th>
<th valign="middle" align="center">Residual connection</th>
<th valign="middle" align="center">Channel reduction ratio</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center">Conv</td>
<td valign="middle" align="center">16</td>
<td valign="middle" align="center">3&#xd7;3</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
</tr>
<tr>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">Conv</td>
<td valign="middle" align="center">32</td>
<td valign="middle" align="center">3&#xd7;3</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
</tr>
<tr>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">C3K2</td>
<td valign="middle" align="center">64</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">0.25</td>
</tr>
<tr>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center">Conv</td>
<td valign="middle" align="center">64</td>
<td valign="middle" align="center">3&#xd7;3</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
</tr>
<tr>
<td valign="middle" align="center">5</td>
<td valign="middle" align="center">C3K2</td>
<td valign="middle" align="center">128</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">&#xd7;</td>
<td valign="middle" align="center">0.25</td>
</tr>
<tr>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">Conv</td>
<td valign="middle" align="center">128</td>
<td valign="middle" align="center">3&#xd7;3</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
</tr>
<tr>
<td valign="middle" align="center">7</td>
<td valign="middle" align="center">C3K2</td>
<td valign="middle" align="center">128</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">/</td>
</tr>
<tr>
<td valign="middle" align="center">8</td>
<td valign="middle" align="center">Conv</td>
<td valign="middle" align="center">256</td>
<td valign="middle" align="center">3&#xd7;3</td>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
</tr>
<tr>
<td valign="middle" align="center">9</td>
<td valign="middle" align="center">C3K2</td>
<td valign="middle" align="center">256</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">/</td>
</tr>
<tr>
<td valign="middle" align="center">10</td>
<td valign="middle" align="center">C2PSA</td>
<td valign="middle" align="center">256</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
</tr>
<tr>
<td valign="middle" align="center">11</td>
<td valign="middle" align="center">Classify</td>
<td valign="middle" align="center">12</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
<td valign="middle" align="center">/</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Additional requirements</title>
<sec id="s5_1">
<label>5.1</label>
<title>Implementation details</title>
<p>In this study, all models were implemented in the PyTorch DL framework and executed on the workstation with a 12th Gen Intel<sup>&#xae;</sup> Core&#x2122; i7-12700F Processor (2.10 GHz), 32.0 GB RAM, and one NVIDIA GeForce RTX 4070 GPU.</p>
<p>During the preprocessing stage, the original sampling rate of the acoustic signal is 16KHz. The cutoff frequency is set to half the sampling rate. Window function selects hamming window. The order of the filter is set to 100. Moreover, the training and testing datasets are split in a ratio of 8:2. Additionally, multiple rounds of 5-fold cross-validation are performed to further assess the effectiveness and robustness of the proposed algorithm across different data partitions. The image size of the input model network is 256*256. The models are trained for 150 epochs on each mini-batch with a batch size of 32. The loss function used in all experiments is cross-entropy. All compared models apply early stopping with a patience of five epochs, whereas our method is trained without early stopping. The hyperparameter settings are as follows: ResNet18 and DenseNet adopt the Adam optimizer, whereas EfficientNet-B0, VGG19, MobileNet, and our proposed model employ SGD. The learning rates are set to 0.001 for ResNet18, DenseNet, and MobileNet; 0.01 for EfficientNet-B0 and VGG19; and 0.1 for our model. Dropout is applied at rates of 0.2 for ResNet18, EfficientNet-B0, and MobileNet, and 0.5 for VGG19. DenseNet and our proposed method do not utilize dropout.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Evaluation</title>
<p>In this study, several evaluation metrics are employed to assess model performance. The confusion matrix analyzes the true versus predicted classifications, providing insight into the types of errors made by the model through the enumeration of true positives, true negatives, false positives and false negatives. Top-1 accuracy (Accuracy@1) measures the proportion of instances where the top predicted class matches the true class, reflecting primary classification accuracy. Macro-Recall quantifies the ratio of true positives to the total number of actual positive instances for each class, then takes the arithmetic mean across all classes. This approach ensures equal evaluation weight for all categories, which is critical when each class holds independent importance. The Macro-F1 score, the harmonic mean of Macro-Precision and Macro-Recall, offers a balanced metric that accounts for the trade-off between these two measures, adopting macro-averaging to ensure uniform assessment of classification consistency. Lastly, the receiver operating characteristic (ROC) curve provides a graphical representation of discriminative power across various classification thresholds, plotting the true positive rate against the false positive rate. The Macro-area under the ROC curve (AUC) serves as an aggregate measure of performance calculated by macro-averaging AUC values across classes, with higher AUC values indicating superior classification ability in maintaining inter-class decision boundary coherence. We deliberately employ macro-averaging to guarantee metric interpretability from a class-agnostic perspective, as this method equally weights the decision patterns of all categories. Furthermore, the number of parameters (in millions) and the computational complexity in GigaFloating Point Operations Per Second (GFLOPS) are used in this study. The number of parameters in the model represents the model&#x2019;s size and capacity, and GFLOPS is a measure of the computational complexity of the model, with higher values indicating more computation is required.</p>
</sec>
</sec>
<sec id="s6" sec-type="results">
<label>6</label>
<title>Results &amp; analysis</title>
<p>In this section, we present the results of classification validation using six models, namely the model used in this paper, Resnet18 (<xref ref-type="bibr" rid="B20">He et&#xa0;al., 2016</xref>) (He Ket al., 2016), Efficientnet-b0 (<xref ref-type="bibr" rid="B44">Tan and Le, 2019</xref>) (Tan M et&#xa0;al., 2019), VGG19 (<xref ref-type="bibr" rid="B42">Simonyan and Zisserman, 2014</xref>) (Simonyan and Zisserman, 2014), MobileNetV2 (<xref ref-type="bibr" rid="B40">Sandler et&#xa0;al., 2018</xref>) (Sandler et&#xa0;al., 2018), DenseNet (<xref ref-type="bibr" rid="B21">Huang et&#xa0;al., 2017</xref>) (Huang G et&#xa0;al., 2017), for different features extracted from the InsectSound1000 dataset. Additionally, we conducted ablation studies and comparisons with other algorithms to further validate the effectiveness and robustness of our proposed method.</p>
<sec id="s6_1">
<label>6.1</label>
<title>Comparative analysis of feature dimensionality</title>
<p>To ensure rigorous benchmarking, the experimental parameters are standardized across all models: the parameter <italic>SegmentFrameNumber Num</italic> is fixed at 10 frames per audio segment and the <italic>samplingrate</italic> is set to 16KHz. As shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, this study demonstrates the superior performance of the proposed PLMS feature and the corresponding model (denoted as Ours in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>) through extensive comparative experiments. Under the PLMS feature representation, our model achieves the highest performance, attaining an Accuracy@1 of 92.4%. Additionally, it achieves a Macro-AUC of 99.70%, significantly surpassing the performance of other models. This underscores the strong discriminative capability of the PLMS feature, which is further leveraged by our model through tailored network design. The superiority of the PLMS feature becomes even more apparent when compared to other feature representations. For instance, within our model, the Accuracy@1 achieved with the PLMS feature is 92.42%, which is 1.61%, 4.34%, 8.5%, 10.13% and 48.97% higher than those achieved with the S-Transform, STFT, wavelet-based feature, Wigner-Ville and LEAF, respectively, with accuracies of 90.96%, 88.58%, 83.92%, 84.92% and 62.04%. This performance improvement can be attributed to the ability of the PLMS feature to integrate the temporal dynamics of pest acoustic signals with multi-scale frequency domain representations, enabling a more comprehensive characterization of audio semantic information. In contrast, baseline features such as S-Transform, STFT, wavelet, and Wigner-Ville predominantly focus on either time-domain or frequency-domain information in isolation. This singular focus limits their capacity to fully capture the complex and multi-faceted nature of audio signals, thereby constraining their generalization capabilities. For instance, S-Transform and STFT primarily emphasize frequency-domain information, while wavelet and Wigner-Ville focus more on time-frequency representations but may not capture the multi-scale characteristics as effectively as the PLMS feature. Although LEAF employs a learnable frontend to adaptively optimize time-frequency representations and enhance robustness, it may still be less effective than PLMS at fully capturing temporal dynamics across multiple scales.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Comparison of classification performance across different audio features and model combinations.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Features</th>
<th valign="middle" align="center">Models</th>
<th valign="middle" align="center">Accuracy@1</th>
<th valign="middle" align="center">Macro-recall</th>
<th valign="middle" align="center">Macro-F1</th>
<th valign="middle" align="center">Macro-AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="6" align="center">PLMS</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">
<bold>92.42</bold>
</td>
<td valign="middle" align="center">
<bold>92.42</bold>
</td>
<td valign="middle" align="center">
<bold>92.41</bold>
</td>
<td valign="middle" align="center">
<bold>99.70</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">87.59</td>
<td valign="middle" align="center">87.59</td>
<td valign="middle" align="center">87.53</td>
<td valign="middle" align="center">97.93</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">88.45</td>
<td valign="middle" align="center">88.45</td>
<td valign="middle" align="center">88.44</td>
<td valign="middle" align="center">99.28</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">8.33</td>
<td valign="middle" align="center">8.33</td>
<td valign="middle" align="center">1.28</td>
<td valign="middle" align="center">50.00</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">84.26</td>
<td valign="middle" align="center">84.26</td>
<td valign="middle" align="center">84.48</td>
<td valign="middle" align="center">98.55</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">77.34</td>
<td valign="middle" align="center">77.34</td>
<td valign="middle" align="center">77.42</td>
<td valign="middle" align="center">97.31</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">S-Transform</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">90.96</td>
<td valign="middle" align="center">90.96</td>
<td valign="middle" align="center">90.99</td>
<td valign="middle" align="center">99.55</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">73.61</td>
<td valign="middle" align="center">73.61</td>
<td valign="middle" align="center">73.71</td>
<td valign="middle" align="center">94.66</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">83.87</td>
<td valign="middle" align="center">83.87</td>
<td valign="middle" align="center">83.82</td>
<td valign="middle" align="center">98.73</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">41.58</td>
<td valign="middle" align="center">41.58</td>
<td valign="middle" align="center">41.70</td>
<td valign="middle" align="center">79.48</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">60.97</td>
<td valign="middle" align="center">61.00</td>
<td valign="middle" align="center">61.57</td>
<td valign="middle" align="center">90.88</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">9.22</td>
<td valign="middle" align="center">9.22</td>
<td valign="middle" align="center">2.45</td>
<td valign="middle" align="center">72.71</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">STFT</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">88.58</td>
<td valign="middle" align="center">88.58</td>
<td valign="middle" align="center">88.63</td>
<td valign="middle" align="center">99.33</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">73.36</td>
<td valign="middle" align="center">73.36</td>
<td valign="middle" align="center">73.13</td>
<td valign="middle" align="center">94.78</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">81.92</td>
<td valign="middle" align="center">81.92</td>
<td valign="middle" align="center">81.91</td>
<td valign="middle" align="center">98.48</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">44.94</td>
<td valign="middle" align="center">44.94</td>
<td valign="middle" align="center">43.97</td>
<td valign="middle" align="center">82.73</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">62.55</td>
<td valign="middle" align="center">62.55</td>
<td valign="middle" align="center">61.10</td>
<td valign="middle" align="center">92.57</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">9.19</td>
<td valign="middle" align="center">9.19</td>
<td valign="middle" align="center">2.32</td>
<td valign="middle" align="center">65.90</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">Wavelet</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">83.92</td>
<td valign="middle" align="center">83.92</td>
<td valign="middle" align="center">83.99</td>
<td valign="middle" align="center">98.69</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">60.65</td>
<td valign="middle" align="center">60.65</td>
<td valign="middle" align="center">60.18</td>
<td valign="middle" align="center">90.09</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">78.37</td>
<td valign="middle" align="center">78.37</td>
<td valign="middle" align="center">78.34</td>
<td valign="middle" align="center">97.94</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">39.38</td>
<td valign="middle" align="center">39.38</td>
<td valign="middle" align="center">39.39</td>
<td valign="middle" align="center">78.06</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">55.97</td>
<td valign="middle" align="center">55.97</td>
<td valign="middle" align="center">55.65</td>
<td valign="middle" align="center">89.05</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">8.55</td>
<td valign="middle" align="center">8.55</td>
<td valign="middle" align="center">1.74</td>
<td valign="middle" align="center">67.51</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">Wigner_ Ville</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">84.92</td>
<td valign="middle" align="center">84.92</td>
<td valign="middle" align="center">84.93</td>
<td valign="middle" align="center">98.94</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">68.47</td>
<td valign="middle" align="center">68.47</td>
<td valign="middle" align="center">68.47</td>
<td valign="middle" align="center">93.04</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">78.89</td>
<td valign="middle" align="center">78.89</td>
<td valign="middle" align="center">78.79</td>
<td valign="middle" align="center">97.94</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">43.50</td>
<td valign="middle" align="center">43.50</td>
<td valign="middle" align="center">42.60</td>
<td valign="middle" align="center">81.59</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">57.40</td>
<td valign="middle" align="center">57.41</td>
<td valign="middle" align="center">57.02</td>
<td valign="middle" align="center">90.15</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">8.33</td>
<td valign="middle" align="center">8.33</td>
<td valign="middle" align="center">1.28</td>
<td valign="middle" align="center">52.30</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">LEAF</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">62.04</td>
<td valign="middle" align="center">62.04</td>
<td valign="middle" align="center">61.62</td>
<td valign="middle" align="center">93.14</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">62.25</td>
<td valign="middle" align="center">62.25</td>
<td valign="middle" align="center">62.28</td>
<td valign="middle" align="center">90.25</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">60.12</td>
<td valign="middle" align="center">60.12</td>
<td valign="middle" align="center">59.89</td>
<td valign="middle" align="center">92.58</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">40.75</td>
<td valign="middle" align="center">40.75</td>
<td valign="middle" align="center">40.31</td>
<td valign="middle" align="center">79.90</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">54.32</td>
<td valign="middle" align="center">53.46</td>
<td valign="middle" align="center">53.27</td>
<td valign="middle" align="center">89.13</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">42.88</td>
<td valign="middle" align="center">42.88</td>
<td valign="middle" align="center">43.34</td>
<td valign="middle" align="center">82.76</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the highest performance achieved by the proposed algorithm among all compared methods for each metric, including Accuracy@1, Macro-recall, Macro-F1, and Macro-AUC.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>From the perspective of model comparison, our model demonstrates a comprehensive advantage under the PLMS feature, achieving an Accuracy@1 of 92.42%, a Macro-F1 score of 92.41%, and a Macro-AUC of 99.70%, significantly outperforming other models. In contrast, ResNet18, although yielding the second-best performance under the PLMS feature with an Accuracy@1 of 87.59%, still trails our model by 4.83%. While EfficientNet-b0 benefits from a compound scaling strategy that enhances computational efficiency, its Accuracy@1 is 3.97% lower than that of our model, highlighting the superior capability of our model to capture long-range temporal dependencies and integrate cross-scale features.</p>
<p>It is noteworthy that traditional deep networks exhibit significant sensitivity to feature representations. For example, VGG19 achieves an Accuracy@1 of only 8.33% and a Macro-F1 of 1.28% under the PLMS feature. This poor performance is largely attributed to its fixed receptive field and redundant parameter design, which are ill-suited for capturing dynamic audio features. DenseNet achieves an Accuracy@1 of only 9.22% under the S-Transform feature, reflecting the limitations of its dense connectivity mechanism in capturing frequency-domain discontinuities. In contrast, MobileNet performs reasonably well with the PLMS feature, achieving an Accuracy@1 of 84.26%. However, its lightweight design compromises its ability to extract deep semantic features, as evidenced by its Accuracy@1 of only 60.97% under the S-Transform feature, considerably lower than the 90.96% achieved by our model. This suggests that, although depthwise separable convolutions are employed in MobileNet to reduce computational overhead, the model&#x2019;s capacity to capture complex acoustic patterns remains limited. Collectively, these findings underscore the importance of the coordinated optimization of the PLMS feature representation and the architectural design of our model as the key determinant of performance improvements.</p>
</sec>
<sec id="s6_2">
<label>6.2</label>
<title>Comparative analysis of various models</title>
<p>As presented in <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>, the number of parameters (in millions) and the computational complexity in GFLOPS of various models under the PLMS feature are systematically compared. From the perspective of parameter quantity, VGG19 has the highest number of parameters, reaching 139.62 million. This substantial parameter volume is a direct consequence of its deep network architecture, making it well-suited for high-precision applications that are less constrained by computational resources. In contrast, ResNet18 contains 11.23 million parameters, while DenseNet has 6.85 million. ResNet18 balances depth and efficiency through residual connections, whereas DenseNet enhances feature reuse via densely connected layers, though at the expense of a larger number of parameters compared to lightweight models. EfficientNet-b0 further reduces parameters to 4.02 million by optimizing parameter utilization through a compound scaling strategy. MobileNet compresses parameters even further to 2.24 million by employing depthwise separable convolutions, showcasing the advantages of lightweight network design. Notably, our model achieves the most aggressive parameter reduction, with only 1.54 million parameters, even surpassing MobileNet. This result underscores its structural innovations in pre-trained model design and fine-tuning strategies.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Comparative analysis of model parameters and computational efficiency under the PLMS feature.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Feature</th>
<th valign="middle" align="center">Models</th>
<th valign="middle" align="center">Param (million)</th>
<th valign="middle" align="center">GFLOPS (GigaFLOPs)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="6" align="center">PLMS</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">1.54M</td>
<td valign="middle" align="center">0.01</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18 (<xref ref-type="bibr" rid="B20">He et&#xa0;al., 2016</xref>)</td>
<td valign="middle" align="center">11.23M</td>
<td valign="middle" align="center">0.07</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0 (<xref ref-type="bibr" rid="B44">Tan and Le, 2019</xref>)</td>
<td valign="middle" align="center">4.02M</td>
<td valign="middle" align="center">0.02</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19 (<xref ref-type="bibr" rid="B42">Simonyan and Zisserman, 2014</xref>)</td>
<td valign="middle" align="center">139.62M</td>
<td valign="middle" align="center">1.04</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet (<xref ref-type="bibr" rid="B40">Sandler et&#xa0;al., 2018</xref>)</td>
<td valign="middle" align="center">2.24M</td>
<td valign="middle" align="center">0.01</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet (<xref ref-type="bibr" rid="B21">Huang et&#xa0;al., 2017</xref>)</td>
<td valign="middle" align="center">6.85M</td>
<td valign="middle" align="center">0.11</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Regarding computational complexity, VGG19 incurs the highest computational cost, with a GFLOPS of 1.04, primarily due to its extensive fully connected layers and deep convolutional operations. ResNet18 and DenseNet maintain moderate computational complexity, with GFLOPS values of 0.07 and 0.11, respectively. While their computational demands are consistent with their parameter volumes, DenseNet incurs additional computational overhead due to its dense connectivity pattern. EfficientNet-b0 significantly reduces computational demands to 0.02 through a well-balanced scaling mechanism. MobileNet and our model achieve the lowest computational complexity, with a GFLOPS of 0.01, highlighting their superior efficiency. This characteristic makes them particularly suitable for real-time processing and low-power environments. In summary, our model emerges as the most lightweight architecture in the comparison, with both the lowest parameter count of 1.54 million and the lowest GFLOPS of 0.01. These results emphasize its advantages in model compression and computational optimization, rendering it highly applicable to scenarios with stringent resource constraints.</p>
</sec>
<sec id="s6_3">
<label>6.3</label>
<title>Comparative analysis of sampling rate dimensionality</title>
<p>To ensure rigorous benchmarking, the experimental parameters <italic>SegmentFrameNumber Num</italic> is fixed at 10 frames per audio segment. The original sampling rate is 16KHz. As shown in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>, we compare classification performance across different sampling rates and model combinations based on the PLMS feature. On one hand, a comprehensive analysis of the dynamic correlation between model performance and sampling rate reveals significant variations among different architectures. Under the PLMS feature, our model exhibits optimal performance at 2,500 Hz, achieving an Accuracy@1 of 96.49%, a Macro-F1 score of 96.49%, and a Macro-AUC of 99.93%. In contrast, VGG19 demonstrates extreme instability at higher sampling rates, particularly at 16 kHz, where its Accuracy@1 plunges to 8.33%, and Macro-F1 drops to 1.28%. This can be attributed to the densely connected fully connected layers in VGG19, which amplify sensitivity to high-resolution noise artifacts. Consequently, the model tends to overfit specific frequency bands, leading to a pronounced degradation in generalization performance. While ResNet18 and EfficientNet-b0 exhibit relatively stable performance across different sampling rates, their performance still shows a slight degradation as the sampling rate decreases. ResNet18 achieves peak performance at 1,500 Hz, attaining an Accuracy@1 of 93.78%, while EfficientNet-b0 performs optimally at 3,000 Hz and 4,000 Hz, reaching an Accuracy@1 of 93.32%. These differences arise from the architectural characteristics of ResNet18 and EfficientNet-b0. ResNet18 utilizes residual connections to effectively mitigate gradient vanishing, enhancing its adaptability to multi-sampling-rate features. In contrast, EfficientNet-b0 employs neural architecture search to optimize the balance between depth, width, and resolution, thereby improving computational efficiency across various sampling rates.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Comparison of classification performance across different sample rates and model combinations based on the PLMS feature.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Sampling rate (Hz)</th>
<th valign="middle" align="center">Models</th>
<th valign="middle" align="center">Accuracy@1</th>
<th valign="middle" align="center">Macro-recall</th>
<th valign="middle" align="center">Macro-F1</th>
<th valign="middle" align="center">Macro-AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="6" align="center">Original</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">92.42</td>
<td valign="middle" align="center">92.42</td>
<td valign="middle" align="center">92.41</td>
<td valign="middle" align="center">99.70</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">87.59</td>
<td valign="middle" align="center">87.59</td>
<td valign="middle" align="center">87.53</td>
<td valign="middle" align="center">97.93</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">88.45</td>
<td valign="middle" align="center">88.45</td>
<td valign="middle" align="center">88.44</td>
<td valign="middle" align="center">99.28</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">8.33</td>
<td valign="middle" align="center">8.33</td>
<td valign="middle" align="center">1.28</td>
<td valign="middle" align="center">50.00</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">84.26</td>
<td valign="middle" align="center">84.26</td>
<td valign="middle" align="center">84.48</td>
<td valign="middle" align="center">98.55</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">77.34</td>
<td valign="middle" align="center">77.34</td>
<td valign="middle" align="center">77.42</td>
<td valign="middle" align="center">97.31</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">4000</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">95.29</td>
<td valign="middle" align="center">95.29</td>
<td valign="middle" align="center">95.29</td>
<td valign="middle" align="center">99.86</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">92.26</td>
<td valign="middle" align="center">92.26</td>
<td valign="middle" align="center">92.26</td>
<td valign="middle" align="center">99.26</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">93.32</td>
<td valign="middle" align="center">93.32</td>
<td valign="middle" align="center">93.33</td>
<td valign="middle" align="center">99.57</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">89.55</td>
<td valign="middle" align="center">89.55</td>
<td valign="middle" align="center">89.60</td>
<td valign="middle" align="center">98.99</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">90.74</td>
<td valign="middle" align="center">90.74</td>
<td valign="middle" align="center">90.74</td>
<td valign="middle" align="center">99.30</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">89.88</td>
<td valign="middle" align="center">89.88</td>
<td valign="middle" align="center">89.92</td>
<td valign="middle" align="center">99.24</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">3500</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">96.03</td>
<td valign="middle" align="center">96.03</td>
<td valign="middle" align="center">96.02</td>
<td valign="middle" align="center">99.87</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">91.60</td>
<td valign="middle" align="center">91.60</td>
<td valign="middle" align="center">91.68</td>
<td valign="middle" align="center">99.22</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">92.20</td>
<td valign="middle" align="center">92.20</td>
<td valign="middle" align="center">92.17</td>
<td valign="middle" align="center">99.45</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">90.48</td>
<td valign="middle" align="center">90.48</td>
<td valign="middle" align="center">90.46</td>
<td valign="middle" align="center">99.25</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">90.01</td>
<td valign="middle" align="center">90.08</td>
<td valign="middle" align="center">90.03</td>
<td valign="middle" align="center">99.21</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">91.20</td>
<td valign="middle" align="center">91.20</td>
<td valign="middle" align="center">91.17</td>
<td valign="middle" align="center">99.40</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">3000</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">95.63</td>
<td valign="middle" align="center">95.63</td>
<td valign="middle" align="center">95.62</td>
<td valign="middle" align="center">99.85</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">93.25</td>
<td valign="middle" align="center">93.25</td>
<td valign="middle" align="center">93.27</td>
<td valign="middle" align="center">99.66</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">93.32</td>
<td valign="middle" align="center">93.32</td>
<td valign="middle" align="center">93.28</td>
<td valign="middle" align="center">99.56</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">91.70</td>
<td valign="middle" align="center">91.07</td>
<td valign="middle" align="center">91.02</td>
<td valign="middle" align="center">99.31</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">88.82</td>
<td valign="middle" align="center">88.82</td>
<td valign="middle" align="center">88.81</td>
<td valign="middle" align="center">99.22</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">89.95</td>
<td valign="middle" align="center">89.95</td>
<td valign="middle" align="center">89.89</td>
<td valign="middle" align="center">99.44</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">2500</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>99.93</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">92.86</td>
<td valign="middle" align="center">92.86</td>
<td valign="middle" align="center">92.84</td>
<td valign="middle" align="center">99.42</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">92.00</td>
<td valign="middle" align="center">92.00</td>
<td valign="middle" align="center">92.01</td>
<td valign="middle" align="center">99.59</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">90.61</td>
<td valign="middle" align="center">90.61</td>
<td valign="middle" align="center">90.61</td>
<td valign="middle" align="center">99.34</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">86.51</td>
<td valign="middle" align="center">86.57</td>
<td valign="middle" align="center">86.81</td>
<td valign="middle" align="center">98.61</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">92.13</td>
<td valign="middle" align="center">92.13</td>
<td valign="middle" align="center">92.11</td>
<td valign="middle" align="center">99.44</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">2000</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">95.21</td>
<td valign="middle" align="center">95.16</td>
<td valign="middle" align="center">95.15</td>
<td valign="middle" align="center">99.82</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">93.12</td>
<td valign="middle" align="center">93.12</td>
<td valign="middle" align="center">93.13</td>
<td valign="middle" align="center">99.43</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">92.33</td>
<td valign="middle" align="center">92.33</td>
<td valign="middle" align="center">92.31</td>
<td valign="middle" align="center">99.54</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">91.01</td>
<td valign="middle" align="center">91.01</td>
<td valign="middle" align="center">91.00</td>
<td valign="middle" align="center">99.29</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">91.67</td>
<td valign="middle" align="center">91.67</td>
<td valign="middle" align="center">91.73</td>
<td valign="middle" align="center">99.51</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">89.42</td>
<td valign="middle" align="center">89.42</td>
<td valign="middle" align="center">89.55</td>
<td valign="middle" align="center">99.21</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">1500</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">96.16</td>
<td valign="middle" align="center">96.16</td>
<td valign="middle" align="center">96.17</td>
<td valign="middle" align="center">99.92</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">93.78</td>
<td valign="middle" align="center">93.78</td>
<td valign="middle" align="center">93.76</td>
<td valign="middle" align="center">99.56</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">92.92</td>
<td valign="middle" align="center">92.92</td>
<td valign="middle" align="center">92.90</td>
<td valign="middle" align="center">99.60</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">89.81</td>
<td valign="middle" align="center">89.81</td>
<td valign="middle" align="center">89.79</td>
<td valign="middle" align="center">99.19</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">88.96</td>
<td valign="middle" align="center">88.96</td>
<td valign="middle" align="center">88.92</td>
<td valign="middle" align="center">99.20</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">91.80</td>
<td valign="middle" align="center">91.80</td>
<td valign="middle" align="center">91.76</td>
<td valign="middle" align="center">99.29</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">1000</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">95.44</td>
<td valign="middle" align="center">95.44</td>
<td valign="middle" align="center">95.45</td>
<td valign="middle" align="center">99.85</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">93.58</td>
<td valign="middle" align="center">93.58</td>
<td valign="middle" align="center">93.57</td>
<td valign="middle" align="center">99.40</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">92.53</td>
<td valign="middle" align="center">92.53</td>
<td valign="middle" align="center">92.49</td>
<td valign="middle" align="center">99.56</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">89.29</td>
<td valign="middle" align="center">89.29</td>
<td valign="middle" align="center">89.23</td>
<td valign="middle" align="center">99.07</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">91.53</td>
<td valign="middle" align="center">91.53</td>
<td valign="middle" align="center">91.51</td>
<td valign="middle" align="center">99.63</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">90.81</td>
<td valign="middle" align="center">90.81</td>
<td valign="middle" align="center">90.84</td>
<td valign="middle" align="center">99.30</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the highest performance achieved by the proposed algorithm among all compared methods for each metric, including Accuracy@1, Macro-recall, Macro-F1, and Macro-AUC.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>In contrast, lightweight models such as MobileNet and DenseNet perform reasonably well at mid-to-low sampling rates but degrade significantly at the original high sampling rate. MobileNet achieves its best Accuracy@1 (91.67%) at 2,000 Hz, while DenseNet peaks at 2,500 Hz with an Accuracy@1 of 92.13%. However, at 16 kHz, the Accuracy@1 of MobileNet drops to 84.26%, while DenseNet declines sharply to 77.34%. This performance gap stems from their distinct architectural constraints. MobileNet employs depthwise separable convolutions, which effectively reduce the number of parameters but also weaken cross-channel dependencies, resulting in suboptimal performance on high-resolution data. In contrast, DenseNet faces increased computational complexity at high sampling rates, where excessive redundant features introduce gradient noise accumulation during backpropagation, ultimately impairing model convergence stability. Notably, despite being designed as a lightweight model, our model demonstrates remarkable cross-sampling-rate stability. By integrating structural innovations, it effectively compensates for the information loss induced by downsampling, thereby enhancing classification robustness under low-sampling-rate conditions. This suggests that lightweight architectures, when appropriately designed, can mitigate performance degradation across varying sampling rates.</p>
<p>One the other hand, examining metric consistency and model reliability further highlights the impact of sampling rates. All models exhibit Macro-AUC values consistently exceeding 99%, indicating strong class discrimination capability. However, discrepancies between Macro-F1 and Accuracy@1 reveal subtle classification biases. For instance, VGG19 achieves a Macro-AUC of 99.29% at 2,000 Hz, yet its Macro-F1 remains lower at 91.00%, suggesting potential class imbalance issues or suboptimal recognition of minority classes. In contrast, our model maintains exceptional consistency across all sampling rates, with Accuracy@1, Macro-F1, and Macro-AUC remaining highly aligned. At 2,500 Hz, it achieves an Accuracy@1 of 96.49%, a Macro-F1 of 96.49%, and a Macro-AUC of 99.93%, confirming its balanced classification capability and low-variance characteristics. These findings underscore that sampling rate has a profound impact on model performance, necessitating a delicate balance between lightweight design and data adaptability. The results suggest that while certain architectures, such as VGG19, struggle with high-resolution noise sensitivity, others, like ResNet18 and EfficientNet-b0, exhibit better adaptability to varying sampling rates. More importantly, our model, through architectural innovation, achieves robust performance at lower sampling rates, making it a highly efficient and practical solution for real-world deployment, particularly in resource-constrained environments where computational efficiency and robustness are crucial.</p>
</sec>
<sec id="s6_4">
<label>6.4</label>
<title>Comparative analysis of patch size</title>
<p>To rigorously analyze the impact of varying patch sizes on the classification performance of different models based on the PLMS, the downsampling rate is set to 2500Hz, as indicated by the results in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>. Subsequently, as delineated in <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref>, we conducted a comparative analysis across four patch sizes. Experimental results demonstrate that patch size has a significant impact on model performance. The proposed model achieves optimal performance when the patch size is configured to a temporal window encompassing 10 successive frames, attaining an Accuracy@1 of 96.49%, a Macro-F1 and Macro-Recall of 96.49%, and a Macro-AUC as high as 99.93%, significantly outperforming the other four models. This finding suggests that the patch size of 10 frames effectively balances local detail and global contextual information. In contrast, a patch size of 5 frames may lack sufficient temporal correlation, leading to incomplete feature representations, while patch sizes of 15 or 20 frames could introduce redundant noise, diminishing the model&#x2019;s sensitivity to informative signals.</p>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Comparison of classification performance of different combinations of patch sizes and models based on the PLMS feature.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Patch size (segmentframenumber)</th>
<th valign="middle" align="center">Models</th>
<th valign="middle" align="center">Accuracy@1</th>
<th valign="middle" align="center">Macro-recall</th>
<th valign="middle" align="center">Macro-F1</th>
<th valign="middle" align="center">Macro-AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="6" align="center">5</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">94.44</td>
<td valign="middle" align="center">94.44</td>
<td valign="middle" align="center">94.43</td>
<td valign="middle" align="center">99.83</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">93.39</td>
<td valign="middle" align="center">93.39</td>
<td valign="middle" align="center">93.40</td>
<td valign="middle" align="center">99.57</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">92.72</td>
<td valign="middle" align="center">92.72</td>
<td valign="middle" align="center">92.74</td>
<td valign="middle" align="center">99.52</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">90.34</td>
<td valign="middle" align="center">90.34</td>
<td valign="middle" align="center">90.35</td>
<td valign="middle" align="center">99.24</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">87.86</td>
<td valign="middle" align="center">87.10</td>
<td valign="middle" align="center">87.15</td>
<td valign="middle" align="center">99.08</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">90.94</td>
<td valign="middle" align="center">90.94</td>
<td valign="middle" align="center">90.90</td>
<td valign="middle" align="center">99.27</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">10</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>99.93</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">92.86</td>
<td valign="middle" align="center">92.86</td>
<td valign="middle" align="center">92.84</td>
<td valign="middle" align="center">99.42</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">92.00</td>
<td valign="middle" align="center">92.00</td>
<td valign="middle" align="center">92.01</td>
<td valign="middle" align="center">99.59</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">90.61</td>
<td valign="middle" align="center">90.61</td>
<td valign="middle" align="center">90.61</td>
<td valign="middle" align="center">99.34</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">86.51</td>
<td valign="middle" align="center">86.57</td>
<td valign="middle" align="center">86.81</td>
<td valign="middle" align="center">98.61</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">92.13</td>
<td valign="middle" align="center">92.13</td>
<td valign="middle" align="center">92.11</td>
<td valign="middle" align="center">99.44</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">15</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">95.50</td>
<td valign="middle" align="center">95.50</td>
<td valign="middle" align="center">95.51</td>
<td valign="middle" align="center">99.84</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">93.65</td>
<td valign="middle" align="center">93.65</td>
<td valign="middle" align="center">93.68</td>
<td valign="middle" align="center">99.54</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">92.33</td>
<td valign="middle" align="center">92.33</td>
<td valign="middle" align="center">92.29</td>
<td valign="middle" align="center">94.48</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">90.41</td>
<td valign="middle" align="center">90.41</td>
<td valign="middle" align="center">90.41</td>
<td valign="middle" align="center">99.36</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">89.04</td>
<td valign="middle" align="center">88.89</td>
<td valign="middle" align="center">88.81</td>
<td valign="middle" align="center">99.27</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">92.26</td>
<td valign="middle" align="center">92.26</td>
<td valign="middle" align="center">92.29</td>
<td valign="middle" align="center">99.59</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">20</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">93.98</td>
<td valign="middle" align="center">93.98</td>
<td valign="middle" align="center">93.97</td>
<td valign="middle" align="center">99.79</td>
</tr>
<tr>
<td valign="middle" align="center">Resnet18</td>
<td valign="middle" align="center">91.34</td>
<td valign="middle" align="center">91.34</td>
<td valign="middle" align="center">91.38</td>
<td valign="middle" align="center">99.25</td>
</tr>
<tr>
<td valign="middle" align="center">Efficientnet-b0</td>
<td valign="middle" align="center">91.01</td>
<td valign="middle" align="center">91.01</td>
<td valign="middle" align="center">90.99</td>
<td valign="middle" align="center">99.34</td>
</tr>
<tr>
<td valign="middle" align="center">VGG19</td>
<td valign="middle" align="center">90.61</td>
<td valign="middle" align="center">90.61</td>
<td valign="middle" align="center">90.60</td>
<td valign="middle" align="center">99.20</td>
</tr>
<tr>
<td valign="middle" align="center">Mobilenet</td>
<td valign="middle" align="center">90.02</td>
<td valign="middle" align="center">89.15</td>
<td valign="middle" align="center">89.36</td>
<td valign="middle" align="center">99.10</td>
</tr>
<tr>
<td valign="middle" align="center">DenseNet</td>
<td valign="middle" align="center">89.35</td>
<td valign="middle" align="center">89.35</td>
<td valign="middle" align="center">89.33</td>
<td valign="middle" align="center">99.38</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the highest performance achieved by the proposed algorithm among all compared methods for each metric, including Accuracy@1, Macro-recall, Macro-F1, and Macro-AUC.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>From the perspective of model-specific characteristics, ResNet18 attains an Accuracy@1 of 92.86% when the patch size is 10 but experiences a decline to 91.34% at a patch size of 20. This decrease arises from interference due to redundant information, underscoring the residual network&#x2019;s proclivity for localized temporal contexts. EfficientNet-b0 maintains relatively stable Accuracy@1 across different patch sizes (ranging from 91.01% to 92.72%). However, when the patch size is 15, it exhibits an anomalous drop in Macro-AUC to 94.48%, which results in spectral confusion among certain categories within the PLMS features. VGG19 exhibits minimal sensitivity to patch size variations, maintaining an Accuracy@1 between 90.34% and 90.61%, as its fixed receptive field and extensive parameterization limit its capacity for capturing dynamic temporal features. MobileNet performs the worst when a patch size is 10 (an Accuracy@1 reaching only 86.51%) but recovers to 89.04% at a patch size of 15. This suggests that its lightweight architecture struggles with modeling medium-length sequences, while longer sequences partially mitigate this limitation through increased information density. DenseNet achieves a peak Accuracy@1 of 92.26% at a patch size of 15 but drops to 89.35% when the patch size reaches 20, likely due to gradient redundancy or noise propagation within its densely connected structure under long-sequence conditions.</p>
<p>In terms of classification metric consistency, the Macro-Recall and Macro-F1 values of all models are highly similar, indicating good inter-class balance in classification results. However, MobileNet exhibits a slight discrepancy at a patch size of 20, with a Macro-Recall of 89.15%, marginally lower than its Macro-F1 score of 89.36%. This suggests reduced sensitivity to certain low-frequency or low-amplitude categories. Notably, EfficientNet-b0 exhibits an anomalous performance at a patch size of 15, with its Macro-AUC decreasing sharply to 94.48%, significantly deviating from the consistently higher Macro-AUC values observed at other patch sizes. This decline suggests that, at this specific patch size, spectral feature confusion occurs among certain categories within longer sequences, thereby impairing the model&#x2019;s discriminative capabilities. Such observations highlight the necessity of jointly modeling local and global features in temporal signal processing to effectively capture both detailed and contextual information.</p>
</sec>
<sec id="s6_5">
<label>6.5</label>
<title>Comparative analysis of confusion matrix and training convergence</title>
<p>
<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> shows the confusion matrices of 12 insect species for the six models based on the PLMS feature, with a patch size of 20 and a downsampling rate of 2,500 Hz. The horizontal axis represents the predicted labels, while the vertical axis represents the true labels, with 0&#x2013;11 corresponding to the insect species listed in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> above. It is worth mentioning that our model exhibits remarkable superiority, as evidenced by its confusion matrix. Except for the 4<sup>th</sup> and 8<sup>th</sup> classes, where the diagonal elements are relatively low, all other classes have diagonal elements approaching 1.0, indicating exceptionally high classification accuracy. Moreover, the off-diagonal elements are minimal, signifying an exceedingly low misclassification rate. In contrast, models such as ResNet18, EfficientNet-B0, VGG19, MobileNet, and DenseNet, while achieving high accuracy in certain categories, exhibit significant misclassifications in specific classes. For instance, ResNet18 shows higher off-diagonal values for the 6<sup>th</sup>, 8<sup>th</sup>, 9<sup>th</sup>, and 11<sup>th</sup> classes; EfficientNet-B0 for the 4<sup>th</sup>, 5<sup>th</sup>, 8<sup>th</sup>, and 11<sup>th</sup> classes; VGG19 for the 4<sup>th</sup>, 5<sup>th</sup>, 8<sup>th</sup>, 10<sup>th</sup>, and 11<sup>th</sup> classes; and DenseNet for the 5<sup>th</sup>, 8<sup>th</sup>, and 11<sup>th</sup> classes. Notably, MobileNet exhibits the diagonal values of approximately 0.6 for the 4<sup>th</sup> and 11<sup>th</sup> classes.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Confusion matrices of the six models based on the PLMS feature.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g008.tif">
<alt-text content-type="machine-generated">Six confusion matrices display classification results with predicted labels on the x-axis and true labels on the y-axis, demonstrating varying accuracy levels. Each matrix uses a color gradient to indicate the frequency of predictions. The diagonal elements represent correct predictions with values close to one, while off-diagonal elements indicate misclassifications with lower values.</alt-text>
</graphic>
</fig>
<p>Focusing on our experimental results, the confusion matrix reveals significant misclassification issues between the 4<sup>th</sup> and 8<sup>th</sup> classes. Samples of the 4<sup>th</sup> class are primarily misclassified as the 5<sup>th</sup> and 11<sup>th</sup> classes, each accounting for 3%, while samples of the 8<sup>th</sup> class are misclassified as the 4<sup>th</sup> and 5<sup>th</sup> classes, each accounting for 2%, and as the 11<sup>th</sup> class, accounting for 3%. Since the data are collected under controlled conditions with minimal background noise, and based on our careful inspection of the spectrograms, the spectral features of the 4<sup>th</sup>, 5<sup>th</sup>, 8<sup>th</sup>, and 11<sup>th</sup> classes exhibit minimal differences. Their frequency distributions and energy concentration regions highly overlap, making traditional spectrogram-based features insufficient to effectively distinguish these classes. Additionally, although the training data quality is high, the diversity of sound samples may be inadequate to capture subtle acoustic variations present in real-world conditions, which limits the model&#x2019;s discriminative power.</p>
<p>To address these challenges, we propose several improvements. First, incorporate richer and more discriminative acoustic features such as transient signal characteristics or nonlinear dynamic features. Second, exploring additional data augmentation techniques, such as pitch shifting and the introduction of simulated environmental effects, may further enhance the diversity of insect acoustic samples and improve the model&#x2019;s generalization ability. Finally, consider multimodal fusion approaches by integrating additional sensory data such as insect vibration signals and behavioral patterns to better differentiate similar classes and reduce misclassification rates between the 4<sup>th</sup> and 8<sup>th</sup> classes.</p>
<p>
<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> shows the validation accuracy and loss progress of all the used six models during the training; each line consists of 150 points, one for each epoch. As illustrated in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, our model rapidly reduces the loss value during the early training stages and achieves high validation accuracy within a relatively small number of iterations, maintaining stability thereafter. This demonstrates fast convergence and strong generalization capability. In comparison, although VGG19 and EfficientNetB0 demonstrate excellent convergence properties, with the loss value of VGG19 approaching zero, their final accuracy remains lower than that achieved by our model. ResNet18 also converges relatively quickly in terms of both loss and accuracy during training, yet its final accuracy still falls short. DenseNet and MobileNet, however, exhibit a slower loss decline and a more gradual accuracy increase during the initial stages of training, followed by noticeable fluctuations in the later phases. Such fluctuations may arise from their network architectures and parameter complexities, which could impede stable training. In contrast, our model excels in training efficiency, accuracy, and stability, clearly demonstrating its advantages.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Validation accuracy and loss progress of the six models based on the PLMS feature.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1610163-g009.tif">
<alt-text content-type="machine-generated">Six line graphs show model training with loss (red line) and accuracy (blue line) over 150 epochs. The top row indicates gradual stabilization of loss and accuracy, while the bottom row displays more significant oscillations in accuracy. Each graph appears to represent different model configurations or datasets.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s6_6">
<label>6.6</label>
<title>Ablation and comparative experiments under cross-validation</title>
<p>As presented in Sections 6.1 to 6.5, a fixed 8:2 dataset split was employed to conduct a baseline performance comparison between our method and representative CNN architectures. While this setting offers a consistent benchmark, it may be influenced by the specific train&#x2013;test partition. To further evaluate model robustness and generalization across diverse data splits, we subsequently performed a five-fold cross-validation experiment, as shown in <xref ref-type="table" rid="T9">
<bold>Table&#xa0;9</bold>
</xref>. In this setting, ablation studies were performed on the proposed method, including its original version, a variant without (w/o) spectral augmentation (SpecAug), and a variant employing patchout, a regularization method that randomly discards a portion of input patches to enhance generalization. These ablation experiments were designed to isolate and quantify the contribution of each component within our framework. In parallel, we carried out comparative experiments with other algorithms, specifically EfficientNet Lite (<xref ref-type="bibr" rid="B41">Sangar and Rajasekar, 2025</xref>) and MobileViT (<xref ref-type="bibr" rid="B18">Gu et&#xa0;al., 2024</xref>), in order to benchmark our method against lightweight architectures proposed in recent literature. Together, these evaluations provide a more comprehensive and rigorous assessment of both accuracy and stability under varying training conditions.</p>
<table-wrap id="T9" position="float">
<label>Table&#xa0;9</label>
<caption>
<p>Ablation and comparative experiments under Cross-Validation.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Data partitioning Strategy</th>
<th valign="middle" align="center">Algorithms</th>
<th valign="middle" align="center">Accuracy@1</th>
<th valign="middle" align="center">Macro-recall</th>
<th valign="middle" align="center">Macro-F1</th>
<th valign="middle" align="center">Macro-AUC</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="5" align="center">Fixed 8:2</td>
<td valign="middle" align="center">
<bold>Ours</bold>
</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>96.49</bold>
</td>
<td valign="middle" align="center">
<bold>99.93</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">
<bold>Ours w/o SpecAug</bold>
</td>
<td valign="middle" align="center">96.03</td>
<td valign="middle" align="center">96.03</td>
<td valign="middle" align="center">96.02</td>
<td valign="middle" align="center">99.92</td>
</tr>
<tr>
<td valign="middle" align="center">
<bold>Ours with Patchout</bold>
</td>
<td valign="middle" align="center">96.23</td>
<td valign="middle" align="center">96.23</td>
<td valign="middle" align="center">96.23</td>
<td valign="middle" align="center">99.90</td>
</tr>
<tr>
<td valign="middle" align="center">(<xref ref-type="bibr" rid="B41">Sangar and Rajasekar, 2025</xref>)<bold>:2025</bold>
</td>
<td valign="middle" align="center">94.25</td>
<td valign="middle" align="center">94.25</td>
<td valign="middle" align="center">94.24</td>
<td valign="middle" align="center">99.79</td>
</tr>
<tr>
<td valign="middle" align="center">(<xref ref-type="bibr" rid="B18">Gu et&#xa0;al., 2024</xref>)<bold>:2024</bold>
</td>
<td valign="middle" align="center">41.17</td>
<td valign="middle" align="center">17.79</td>
<td valign="middle" align="center">15.25</td>
<td valign="middle" align="center">69.64</td>
</tr>
<tr>
<td valign="middle" rowspan="5" align="center">Five-fold<break/>cross-validation</td>
<td valign="middle" align="center">
<bold>Ours</bold>
</td>
<td valign="middle" align="center">
<bold>95.23&#xb1;0.17</bold>
</td>
<td valign="middle" align="center">
<bold>95.23&#xb1;0.17</bold>
</td>
<td valign="middle" align="center">
<bold>95.23&#xb1; 0.17</bold>
</td>
<td valign="middle" align="center">
<bold>99.85&#xb1;0.04</bold>
</td>
</tr>
<tr>
<td valign="middle" align="center">
<bold>Ours w/o SpecAug</bold>
</td>
<td valign="middle" align="center">95.04&#xb1;0.60</td>
<td valign="middle" align="center">95.04&#xb1;0.60</td>
<td valign="middle" align="center">95.04&#xb1; 0.60</td>
<td valign="middle" align="center">99.84&#xb1;0.04</td>
</tr>
<tr>
<td valign="middle" align="center">
<bold>Ours with Patchout</bold>
</td>
<td valign="middle" align="center">94.65&#xb1;0.30</td>
<td valign="middle" align="center">94.65&#xb1;0.30</td>
<td valign="middle" align="center">94.65&#xb1; 0.30</td>
<td valign="middle" align="center">99.85&#xb1;0.02</td>
</tr>
<tr>
<td valign="middle" align="center">(<xref ref-type="bibr" rid="B41">Sangar and Rajasekar, 2025</xref>)<bold>:2025</bold>
</td>
<td valign="middle" align="center">91.46&#xb1;0.75</td>
<td valign="middle" align="center">91.41&#xb1;0.75</td>
<td valign="middle" align="center">91.43&#xb1; 0.75</td>
<td valign="middle" align="center">99.58&#xb1;0.06</td>
</tr>
<tr>
<td valign="middle" align="center">(<xref ref-type="bibr" rid="B18">Gu et&#xa0;al., 2024</xref>)<bold>:2024</bold>
</td>
<td valign="middle" align="center">34.69&#xb1;12.33</td>
<td valign="middle" align="center">26.23&#xb1; 10.89</td>
<td valign="middle" align="center">22.4610.92</td>
<td valign="middle" align="center">73.34&#xb1;8.66</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold values indicate the highest performance achieved by the proposed algorithm among all compared methods for each metric, including Accuracy@1, Macro-recall, Macro-F1, and Macro-AUC.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The results shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> indicate that, regarding the fixed 8:2 dataset partition, our model attains the highest performance across all evaluation metrics. In contrast, the variants without SpecAug or incorporating Patchout show a marginal performance decline but still substantially outperform the baseline methods reported in references (<xref ref-type="bibr" rid="B41">Sangar and Rajasekar, 2025</xref>) and (<xref ref-type="bibr" rid="B18">Gu et&#xa0;al., 2024</xref>). Notably, the method in (<xref ref-type="bibr" rid="B18">Gu et&#xa0;al., 2024</xref>) exhibits significantly inferior results, with all metrics falling well below those of our model. From the perspective of five-fold cross-validation, the robustness of our model is further confirmed, consistently leading with an accuracy of 95.23% and a minimal standard deviation of 0.17, underscoring its strong generalization capability. The variant without SpecAug and the one incorporating Patchout maintain stable performance, although with slightly reduced scores. Conversely, the methods from (<xref ref-type="bibr" rid="B41">Sangar and Rajasekar, 2025</xref>) and (<xref ref-type="bibr" rid="B18">Gu et&#xa0;al., 2024</xref>) demonstrate comparatively poorer and more variable outcomes, particularly (<xref ref-type="bibr" rid="B18">Gu et&#xa0;al., 2024</xref>), which shows a markedly high standard deviation, indicating less stable performance. In summary, the proposed approach not only achieves superior accuracy on the fixed dataset partition, but also demonstrates exceptional stability and generalization under the more rigorous five-fold cross-validation, thereby validating the effectiveness of the model architecture and data augmentation strategies employed.</p>
</sec>
</sec>
<sec id="s7" sec-type="conclusions">
<label>7</label>
<title>Conclusions</title>
<p>Early detection of pest infestations is critical for mitigating the adverse effects on agricultural productivity and ensuring ecological balance. Reviewed studies highlight the importance of balancing accuracy, scalability, and robustness in pest detection. Building upon these insights, this study presents a novel cross-modal adaptation approach for early-stage pest surveillance, utilizing the comprehensive bioacoustic InsectSound1000 database. By employing adaptive audio preprocessing, the approach effectively filters high-frequency noise and reduces computational complexity through downsampling. The utilization of PLMS spectrograms facilitates the refined transformation of acoustic signals into visual representations, enhancing the precision of time-frequency pattern extraction. The deployment of the YOLOv11 model for deep transfer learning enables the extraction of high-level features, thereby enhancing precision and the ability to generalize across diverse datasets. Experimental results demonstrate that the proposed method achieves high detection accuracy while maintaining manageable computational complexity. This framework offers a promising alternative to conventional pest monitoring techniques, paving the way for integrated, automated pest management systems that combine acoustic and visual modalities for enhanced early surveillance and pest control. However, since the InsectSound1000 dataset is collected under controlled conditions, it does not comprehensively represent practical challenges such as hardware dependency, the requirement for specialized equipment, and environmental noise encountered in real-world agricultural environments. To address these limitations, we are actively conducting field surveys and on-site experiments aimed at further validating and optimizing the proposed method for effective deployment in operational settings. Future research could focus on expanding the dataset to include a broader range of insect species and environmental conditions, which could further enhance the robustness of the model. Additionally, integrating the proposed system into existing pest management frameworks would enable automated, real-time surveillance, further optimizing pest control strategies.</p>
</sec>
</body>
<back>
<sec id="s8" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding author/s.</p>
</sec>
<sec id="s9" sec-type="author-contributions">
<title>Author contributions</title>
<p>YW: Writing &#x2013; review &amp; editing, Writing &#x2013; original draft, Formal analysis, Software, Visualization, Methodology, Data curation, Conceptualization, Validation, Investigation. LC: Data curation, Writing &#x2013; review &amp; editing, Software. WH: Methodology, Investigation, Formal analysis, Writing &#x2013; review &amp; editing. JN: Software, Writing &#x2013; review &amp; editing, Data curation. YL: Supervision, Writing &#x2013; review &amp; editing. YC: Writing &#x2013; review &amp; editing, Investigation. JL: Resources, Writing &#x2013; review &amp; editing, Project administration. XL: Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s10" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This work was supported by the Shanxi Province Basic Research Program (202203021212444); Shanxi Agricultural University doctoral research startup project (2021BQ88); Shanxi Higher Education Science and Technology Innovation Project (2024L058); Shanxi Provincial Applied Basic Research Project (202203021212455); Shanxi Agricultural University 24-School Outstanding Doctoral Startup Project (6K245406008).</p>
</sec>
<sec id="s11" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s12" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s13" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ali</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Dhanaraj</surname> <given-names>R. K.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Heterogeneous features and deep learning networks fusion-based pest detection, prevention and controlling system using IoT and pest sound analytics in a vast agriculture system</article-title>. <source>Comput. Electrical Eng.</source> <volume>116</volume>, <fpage>109146</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compeleceng.2024.109146</pub-id>
</citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z. Y.</given-names>
</name>
<name>
<surname>Wang</surname>
</name>
</person-group> (<year>2024</year>). <article-title>Impact of climate change on agricultural productivity: a combination of spatial Durbin model and entropy approaches</article-title>. <source>Int. J. Climate Change Strategies Manage.</source> <volume>16</volume>, <fpage>26</fpage>&#x2013;<lpage>48</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1108/IJCCSM-02-2022-0016</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balingbing</surname> <given-names>C. B.</given-names>
</name>
<name>
<surname>Kirchner</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Siebald</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Kaufmann</surname> <given-names>H. H.</given-names>
</name>
<name>
<surname>Gummert</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Van Hung</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Application of a multi-layer convolutional neural network model to classify major insect pests in stored rice detected by an acoustic device</article-title>. <source>Comput. Electron. Agric.</source> <volume>225</volume>, <fpage>109297</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109297</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Basak</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ghosh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Saha</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Dutta</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Maiti</surname> <given-names>A.</given-names>
</name>
</person-group>. (<year>2022</year>). &#x201c;<article-title>Insects sound classification with acoustic features and k-nearest algorithm</article-title>,&#x201d; in <conf-name>PREPARE@ u&#xae;| FOSET Conferences</conf-name>. <publisher-name>CALNESTOR Knowledge Solutions Pvt. Ltd.</publisher-name>, <publisher-loc>Hyderabad, India.</publisher-loc>
</citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boulila</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Alzahem</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Koubaa</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Benjdira</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ammar</surname> <given-names>A.</given-names>
</name>
</person-group>. (<year>2023</year>). <article-title>Early detection of red palm weevil infestations using deep learning classification of acoustic signals</article-title>. <source>Comput. Electron. Agric.</source> <volume>212</volume>, <fpage>108154</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108154</pub-id>
</citation></ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Branding</surname> <given-names>J.</given-names>
</name>
<name>
<surname>von H&#xf6;rsten</surname> <given-names>D.</given-names>
</name>
<name>
<surname>B&#xf6;ckmann</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Wegener</surname> <given-names>J. K.</given-names>
</name>
<name>
<surname>Hartung</surname> <given-names>E.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Dataset: Insectsound1000</source> (<publisher-loc>G&#xf6;ttingen, Germany</publisher-loc>: <publisher-name>OpenAgrar Repository</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.5073/20231024-173119-0</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deguine</surname> <given-names>J. P.</given-names>
</name>
<name>
<surname>Aubertot</surname> <given-names>J. N.</given-names>
</name>
<name>
<surname>Flor</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Lescourret</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Wyckhuys</surname> <given-names>K. A. G.</given-names>
</name>
<name>
<surname>Ratnadass</surname> <given-names>A</given-names>
</name>
</person-group>. (<year>2021</year>). <article-title>Integrated pest management: good intentions, hard realities. A review</article-title>. <source>Agron. Sustain. Dev.</source> <volume>41</volume>, <fpage>38</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s13593-021-00689-w</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Souza</surname> <given-names>U. B.</given-names>
</name>
<name>
<surname>Escola</surname> <given-names>J. P. L.</given-names>
</name>
<name>
<surname>Maccagnan</surname> <given-names>D. H. B.</given-names>
</name>
<name>
<surname>Brito</surname> <given-names>L. d. C.</given-names>
</name>
<name>
<surname>Guido</surname> <given-names>R. C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Empirical mode decomposition applied to acoustic detection of a cicadid pest</article-title>. <source>Comput. Electron. Agric.</source> <volume>199</volume>, <fpage>107181</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.107181</pub-id>
</citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dhanaraj</surname> <given-names>R. K.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Nayyar</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Deep Multibranch Fusion Residual Network and IoT-based pest detection system using sound analytics in large agricultural field</article-title>. <source>Multimedia Tools Appl.</source> <volume>83</volume> (<issue>13</issue>), <fpage>40215</fpage>&#x2013;<lpage>40252</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042-023-16897-3</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dong</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Insect sound recognition based on convolutional neural network</article-title>,&#x201d; in <conf-name>2018 IEEE 3rd international conference on image, vision and computing (ICIVC)</conf-name>. <fpage>855</fpage>&#x2013;<lpage>859</lpage>, <publisher-name>IEEE</publisher-name>, <publisher-loc>Piscataway, NJ, USA</publisher-loc>.</citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Escola</surname> <given-names>J. P. L.</given-names>
</name>
<name>
<surname>Guido</surname> <given-names>R. C.</given-names>
</name>
<name>
<surname>da Silva</surname> <given-names>I. N.</given-names>
</name>
<name>
<surname>Cardoso</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Maccagnan</surname> <given-names>D. H. B.</given-names>
</name>
<name>
<surname>Dezotti</surname> <given-names>A. K</given-names>
</name>
</person-group>. (<year>2020</year>). <article-title>Automated acoustic detection of a cicadid pest in coffee plantations</article-title>. <source>Comput. Electron. Agric.</source> <volume>169</volume>, <fpage>105215</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105215</pub-id>
</citation></ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Fai&#xdf;</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <source>InsectSet32: Dataset for automatic acoustic identification of insects (Orthoptera and Cicadidae)</source> (<publisher-loc>Geneva, Switzerland</publisher-loc>: <publisher-name>Zenodo</publisher-name>).</citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fai&#xdf;</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Stowell</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Adaptive representations of sound for automatic insect recognition</article-title>. <source>PloS Comput. Biol.</source> <volume>19</volume>, <elocation-id>e1011541</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pcbi.1011541</pub-id>, PMID: <pub-id pub-id-type="pmid">37792895</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<collab>FAO</collab>
<collab>IFAD</collab>
<collab>UNICEF</collab>
<collab>WFP</collab>
<collab>WHO</collab>
</person-group>. (<year>2024</year>). <source>The State of Food Security and Nutrition in the World 2024</source>. <publisher-name>FAO</publisher-name>: <publisher-loc>Rome, Italy</publisher-loc>, 2024.</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferreira</surname> <given-names>A. I. S.</given-names>
</name>
<name>
<surname>da Silva</surname> <given-names>N. F. F.</given-names>
</name>
<name>
<surname>Mesquita</surname> <given-names>F. N.</given-names>
</name>
<name>
<surname>Rosa</surname> <given-names>T. C.</given-names>
</name>
<name>
<surname>Buchmann</surname> <given-names>S. L.</given-names>
</name>
<name>
<surname>Mesquita-Neto</surname> <given-names>J. N.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Transformer Models improve the acoustic recognition of buzz-pollinating bee species</article-title>. <source>Ecol. Inf.</source> <volume>86</volume>, <fpage>103010</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2025.103010</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gandara</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Jacoby</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Laurent</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Spatuzzi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vlachopoulos</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Borst</surname> <given-names>N. O.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Pervasive sublethal effects of agrochemicals on insects at environmentally relevant concentrations</article-title>. <source>Science</source> <volume>386</volume>, <fpage>446</fpage>&#x2013;<lpage>453</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.ado0251</pub-id>, PMID: <pub-id pub-id-type="pmid">39446951</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ghosh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Biswas</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Exponential population growth and global food security: challenges and alternatives</article-title>,&#x201d; in <source>Bioremediation of Emerging Contaminants from Soils</source> (<publisher-loc>Amsterdam, Netherlands</publisher-loc>: <publisher-name>Elsevier</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>20</lpage>.</citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>You</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An adaptive acoustic signal reconstruction and fault diagnosis method for rolling bearings based on SSDAE&#x2013;MobileViT</article-title>. <source>Measurement Sci. Technol.</source> <volume>36</volume>, <fpage>016190</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1088/1361-6501/ad98b1</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhaopeng</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Enhancing insect sound classification using dual-tower network: A fusion of temporal and spectral feature perception</article-title>. <source>Appl. Sci.</source> <volume>14</volume>, <fpage>3116</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app14073116</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J</given-names>
</name>
</person-group>. (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <fpage>770</fpage>&#x2013;<lpage>778</lpage>. <publisher-name>IEEE</publisher-name>, <publisher-loc>Piscataway, NJ, USA</publisher-loc>
</citation></ref>
<ref id="B21">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>van der Maaten</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Weinberger</surname> <given-names>K. Q.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Densely connected convolutional networks</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <fpage>4700</fpage>&#x2013;<lpage>4708</lpage>. <publisher-name>IEEE,</publisher-name> <publisher-loc>Piscataway, NJ, USA.</publisher-loc>
</citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karar</surname> <given-names>M. E.</given-names>
</name>
<name>
<surname>Reyad</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Abdel-Aty</surname> <given-names>A. H.</given-names>
</name>
<name>
<surname>Owyed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hassan</surname> <given-names>M. F.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Intelligent IoT-aided early sound detection of red palm weevils</article-title>. <source>Cmc-Comput. Mater. Contin</source> <volume>69</volume>, <fpage>4095</fpage>&#x2013;<lpage>4111</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.32604/cmc.2021.019059</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kiobia</surname> <given-names>D. O.</given-names>
</name>
<name>
<surname>Mwitta</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Fue</surname> <given-names>K. G.</given-names>
</name>
<name>
<surname>Schmidt</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Riley</surname> <given-names>D. G.</given-names>
</name>
<name>
<surname>Rains</surname> <given-names>G. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A review of successes and impeding challenges of IoT-based insect pest detection systems for estimating agroecosystem health and productivity of cotton</article-title>. <source>Sensors</source> <volume>23</volume>, <fpage>4127</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s23084127</pub-id>, PMID: <pub-id pub-id-type="pmid">37112469</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kulyukin</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Mukherjee</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Amlathe</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Toward audio beehive monitoring: deep learning vs. standard machine learning in classifying beehive audio samples</article-title>. <source>Appl. Sci.</source> <volume>8</volume>, <elocation-id>1573</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app8091573</pub-id>
</citation></ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Le-Qing</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2011</year>). &#x201c;<article-title>Insect sound recognition based on mfcc and pnn</article-title>,&#x201d; in <conf-name>2011 International Conference on Multimedia and Signal Processing</conf-name>, Vol. <volume>2</volume>. <fpage>42</fpage>&#x2013;<lpage>46</lpage>,<publisher-name> IEEE</publisher-name>, <publisher-loc>Piscataway, NJ, USA</publisher-loc>.</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X</given-names>
</name>
</person-group>. (<year>2021</year>). <article-title>Classification and detection of insects from field images using deep learning for smart pest management: A systematic review</article-title>. <source>Ecol. Inf.</source> <volume>66</volume>, <fpage>101460</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2021.101460</pub-id>
</citation></ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Ni</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Key contributions of the overexpressed Plutella xylostella sigma glutathione S-transferase 1 Gene (PxGSTs1) in the resistance evolution to multiple insecticides</article-title>. <source>J. Agric. Food Chem.</source> <volume>72</volume>, <fpage>2560</fpage>&#x2013;<lpage>2572</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acs.jafc.3c09458</pub-id>, PMID: <pub-id pub-id-type="pmid">38261632</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Co3O4/CuO@ C catalyst based on cobalt-doped HKUST-1 as an efficient peroxymonosulfate activator for pendimethalin degradation: Catalysis and mechanism</article-title>. <source>J. Hazardous Materials</source> <volume>478</volume>, <fpage>135437</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jhazmat.2024.135437</pub-id>, PMID: <pub-id pub-id-type="pmid">39121735</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mankin</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Bug Bytes sound library: stored product insect pest sounds. Ag Data Commons</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.15482/USDA.ADC/1504600</pub-id>
</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mankin</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Hagstrum</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Eliopoulos</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Njoroge</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Automated applications of acoustics for stored product insect detection, monitoring, and management</article-title>. <source>Insects</source> <volume>12</volume>, <fpage>259</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/insects12030259</pub-id>, PMID: <pub-id pub-id-type="pmid">33808747</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Marshall</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Hill</surname> <given-names>K.</given-names>
</name>
</person-group> <article-title>Insectsingers</article-title>. Available online at: <uri xlink:href="http://www.insectsingers.com/">http://www.insectsingers.com/</uri> (Accessed <access-date>April 23 2019</access-date>).</citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Montemayor</surname> <given-names>J. J. M.</given-names>
</name>
<name>
<surname>Escuadra</surname> <given-names>G. P. G.</given-names>
</name>
<name>
<surname>Nambatac</surname> <given-names>M. A. G.</given-names>
</name>
<name>
<surname>Tenoria</surname> <given-names>D. T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Detecting rice weevils in stored grains using MFCC and CNN</article-title>. <source>Proc. Comput. Sci.</source> <volume>234</volume>, <fpage>1681</fpage>&#x2013;<lpage>1688</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.procs.2024.03.173</pub-id>
</citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ngugi</surname> <given-names>L. C.</given-names>
</name>
<name>
<surname>Abelwahab</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Abo-Zahhad</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Recent advances in image processing techniques for automated leaf pest and disease recognition&#x2013;A review</article-title>. <source>Inf. Process. Agric.</source> <volume>8</volume>, <fpage>27</fpage>&#x2013;<lpage>51</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.inpa.2020.04.004</pub-id>
</citation></ref>
<ref id="B34">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Noda</surname> <given-names>J. J.</given-names>
</name>
<name>
<surname>Travieso</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Rodr&#xed;guez</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Dutta</surname> <given-names>M. K.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Using bioacoustic signals and support vector machine for automatic classification of insects</article-title>,&#x201d; in <conf-name>2016 3rd international conference on signal processing and integrated networks (SPIN)</conf-name>. <fpage>656</fpage>&#x2013;<lpage>659</lpage>, <publisher-name>IEEE</publisher-name>, <publisher-loc>Piscataway, NJ, USA</publisher-loc>.</citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Noda</surname> <given-names>J. J.</given-names>
</name>
<name>
<surname>Travieso-Gonz&#xe1;lez</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>S&#xe1;nchez-Rodr&#xed;guez</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Alonso-Hern&#xe1;ndez</surname> <given-names>J. B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Acoustic classification of singing insects based on MFCC/LFCC fusion</article-title>. <source>Appl. Sci.</source> <volume>9</volume>, <fpage>4097</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app9194097</pub-id>
</citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Phan</surname> <given-names>T. T. H.</given-names>
</name>
<name>
<surname>Nguyen-Doan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Nguyen-Huu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Nguyen-Van</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Pham-Hong</surname> <given-names>T</given-names>
</name>
</person-group>. (<year>2023</year>). <article-title>Investigation on new Mel frequency cepstral coefficients features and hyper-parameters tuning technique for bee sound recognition</article-title>. <source>Soft Computing</source> <volume>27</volume>, <fpage>5873</fpage>&#x2013;<lpage>5892</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00500-022-07596-6</pub-id>
</citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Phung</surname> <given-names>Q. V.</given-names>
</name>
<name>
<surname>Ahmad</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Habibi</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Hinckley</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Automated insect detection using acoustic features based on sound generated from insect activities</article-title>. <source>Acoustics Aust.</source> <volume>45</volume>, <fpage>445</fpage>&#x2013;<lpage>451</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s40857-017-0095-6</pub-id>
</citation></ref>
<ref id="B38">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Piczak</surname> <given-names>K. J.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>ESC: dataset for environmental sound classification</article-title>,&#x201d; in <conf-name>Proceedings of the 23rd Annual ACM Conference on Multimedia</conf-name>, <conf-loc>Brisbane, Australia. ACM, New York, NY, USA</conf-loc>.</citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rusdiyana</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Sutrisno</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Harsono</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A bibliometric review of sustainable agriculture in rural development</article-title>. <source>West Sci. Interdiscip. Stud.</source> <volume>2</volume>, <fpage>630</fpage>&#x2013;<lpage>637</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.58812/wsis.v2i03.747</pub-id>
</citation></ref>
<ref id="B40">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sandler</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Howard</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zhmoginov</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>L. C</given-names>
</name>
</person-group>. (<year>2018</year>). &#x201c;<article-title>Mobilenetv2: Inverted residuals and linear bottlenecks</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>. <fpage>4510</fpage>&#x2013;<lpage>4520</lpage>. <publisher-name>IEEE</publisher-name>, <publisher-loc>Piscataway, NJ, USA</publisher-loc>.</citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sangar</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Rajasekar</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Optimized classification of potato leaf disease using EfficientNet-LITE and KE-SVM in diverse environments</article-title>. <source>Front. Plant Sci.</source> <volume>16</volume>, <elocation-id>1499909</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2025.1499909</pub-id>, PMID: <pub-id pub-id-type="pmid">40385236</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zisserman</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arxiv preprint arxiv:1409.1556</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id>
</citation></ref>
<ref id="B43">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Szekeres</surname> <given-names>B. J.</given-names>
</name>
<name>
<surname>Gy&#xf6;ngy&#xf6;ssy</surname> <given-names>M. N.</given-names>
</name>
<name>
<surname>Botzheim</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>A resnet-9 model for insect wingbeat sound classification</article-title>,&#x201d; in <conf-name>2023 IEEE Symposium Series on Computational Intelligence (SSCI)</conf-name>. <fpage>587</fpage>&#x2013;<lpage>592</lpage>, <publisher-name>IEEE</publisher-name>, <publisher-loc>Piscataway, NJ, USA</publisher-loc>.</citation></ref>
<ref id="B44">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Le</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Efficientnet: Rethinking model scaling for convolutional neural networks</article-title>,&#x201d; in <conf-name>International conference on machine learning</conf-name>. <fpage>6105</fpage>&#x2013;<lpage>6114</lpage>, <publisher-name>PMLR</publisher-name>, <publisher-loc>Vienna, Austria</publisher-loc>.</citation></ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tey</surname> <given-names>W. T.</given-names>
</name>
<name>
<surname>Connie</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Choo</surname> <given-names>K. Y.</given-names>
</name>
<name>
<surname>Goh</surname> <given-names>M. K. O.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Cicada species recognition based on acoustic signals</article-title>. <source>Algorithms</source> <volume>15</volume>, <fpage>358</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/a15100358</pub-id>
</citation></ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Truong</surname> <given-names>T. H.</given-names>
</name>
<name>
<surname>Du Nguyen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Mai</surname> <given-names>T. Q. A.</given-names>
</name>
<name>
<surname>Nguyen</surname> <given-names>H. L.</given-names>
</name>
<name>
<surname>Dang</surname> <given-names>T. N. M.</given-names>
</name>
<name>
<surname>Phan</surname> <given-names>T.-T.-H.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>A deep learning-based approach for bee sound identification</article-title>. <source>Ecol. Inf.</source> <volume>78</volume>, <fpage>102274</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2023.102274</pub-id>
</citation></ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Varzakas</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Smaoui</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Global food security and sustainability issues: the road to 2030 from nutrition and sustainable healthy diets to food systems change</article-title>. <source>Foods</source> <volume>13</volume>, <fpage>306</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/foods13020306</pub-id>, PMID: <pub-id pub-id-type="pmid">38254606</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Walker</surname> <given-names>T. J.</given-names>
</name>
<name>
<surname>Moore</surname> <given-names>T. E</given-names>
</name>
</person-group> (<year>2019</year>). <source>Singing Insects of North America(SINA) Collection</source> (<publisher-name>University of Florida</publisher-name>). Available online at: <uri xlink:href="http://entnemdept.ufl.edu/walker/buzz/">http://entnemdept.ufl.edu/walker/buzz/</uri> (<access-date>Accessed on 24 April 2019</access-date>).</citation></ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Vhaduri</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Sound classification of four insect classes</article-title>. <source>arXiv preprint arXiv:2412.12395</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2412.12395</pub-id>
</citation></ref>
<ref id="B50">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>R. R.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>PEDS-AI: A novel unmanned aerial vehicle based artificial intelligence powered visual-acoustic pest early detection and identification system for field deployment and surveillance</article-title>,&#x201d; in <conf-name>2023 IEEE Conference on Technologies for Sustainability (SusTech)</conf-name>. <fpage>12</fpage>&#x2013;<lpage>19</lpage>, <publisher-name>IEEE</publisher-name>, <publisher-loc>Piscataway, NJ, USA</publisher-loc>.</citation></ref>
<ref id="B51">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>A novel insect sound recognition algorithm based on MFCC and CNN</article-title>,&#x201d; in <conf-name>2021 6th International Conference on Communication, Image and Signal Processing (CCISP)</conf-name>. <fpage>289</fpage>&#x2013;<lpage>294</lpage>, <publisher-name>IEEE</publisher-name>, <publisher-loc>Piscataway, NJ, USA</publisher-loc>.</citation></ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Arcot</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Medina</surname> <given-names>R. F.</given-names>
</name>
<name>
<surname>Bernal</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cisneros-Zevallos</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Akbulut</surname> <given-names>M. E. S</given-names>
</name>
</person-group>. (<year>2024</year>). <article-title>Integrated pest management: an update on the sustainability approach to crop protection</article-title>. <source>ACS omega</source> <volume>9</volume>, <fpage>41130</fpage>&#x2013;<lpage>41147</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acsomega.4c06628</pub-id>, PMID: <pub-id pub-id-type="pmid">39398119</pub-id></citation></ref>
</ref-list>
</back>
</article>