<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1522510</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Few-shot object detection for pest insects via features aggregation and contrastive learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>He</surname>
<given-names>Shuqian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2909273/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jin</surname>
<given-names>Biao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sun</surname>
<given-names>Xuechao</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Jiang</surname>
<given-names>Wenjuan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2862896/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gu</surname>
<given-names>Jiaxing</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gu</surname>
<given-names>Fenglin</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/532040/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Information Science and Technology, Hainan Normal University</institution>, <addr-line>Haikou, Hainan</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Hainan Provincial Engineering Research Center for Artificial Intelligence and Equipment for Monitoring Tropical Biodiversity and Ecological Environment, Hainan Normal University</institution>, <addr-line>Haikou, Hainan</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>College of Computer Science and Technology, Zhejiang University</institution>, <addr-line>Hangzhou, Zhejiang</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Spice and Beverage Research Institute, Chinese Academy of Tropical Agricultural Sciences</institution>, <addr-line>Wanning, Hainan</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Lei Shu, Nanjing Agricultural University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Yalin Wu, Peking University, China</p>
<p>Gurminder Singh, North Dakota State University, United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Shuqian He, <email xlink:href="mailto:hsq@hainnu.edu.cn">hsq@hainnu.edu.cn</email>; Wenjuan Jiang, <email xlink:href="mailto:jwj@hainnu.edu.cn">jwj@hainnu.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>06</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1522510</elocation-id>
<history>
<date date-type="received">
<day>13</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>05</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 He, Jin, Sun, Jiang, Gu and Gu</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>He, Jin, Sun, Jiang, Gu and Gu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Accurate detection of pest insects is critical for agricultural pest management and crop yield protection, yet traditional detection methods struggle due to the vast diversity of pest species, significant individual differences, and limited labeled data. These challenges are compounded by the typically small size of pest targets and complex environmental conditions. To address these limitations, this study proposes a novel few-shot object detection (FSOD) method leveraging feature aggregation and supervised contrastive learning (SCL) within the Faster R-CNN framework. Our methodology involves multi-scale feature extraction using a Feature Pyramid Network (FPN), enabling the capture of rich semantic information across various scales. A Feature Aggregation Module (FAM) with an attention mechanism is designed to effectively fuse contextual features from support and query images, enhancing representation capabilities for multi-scale and few-sample pest targets. Additionally, supervised contrastive learning is employed to strengthen intra-class similarity and inter-class dissimilarity, thereby improving discriminative power. To manage class imbalance and enhance the focus on challenging samples, focal loss and class weights are integrated into the model&#x2019;s comprehensive loss function. Experimental validation on the PestDet20 dataset, consisting of diverse tropical pest insects, demonstrates that the proposed method significantly outperforms existing approaches, including YOLO, TFA, VFA, and FSCE. Specifically, our model achieves superior mean Average Precision (mAP) results across different few-shot scenarios (3-shot, 5-shot, and 10-shot), demonstrating robustness and stability. Ablation studies confirm that each component of our method substantially contributes to performance improvement. This research provides a practical and efficient solution for pest detection under challenging conditions, reducing dependency on large annotated datasets and improving detection accuracy for minority pest classes. While computational complexity remains higher than real-time frameworks like YOLO, the significant gains in detection accuracy justify the trade-off for critical pest management applications.</p>
</abstract>
<kwd-group>
<kwd>feature aggregation</kwd>
<kwd>contrastive learning</kwd>
<kwd>few-shot learning</kwd>
<kwd>object detection</kwd>
<kwd>pest control</kwd>
</kwd-group>
<counts>
<fig-count count="11"/>
<table-count count="8"/>
<equation-count count="21"/>
<ref-count count="49"/>
<page-count count="23"/>
<word-count count="12623"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Accurate detection of pest insects is crucial for effective pest management and agricultural productivity. Pest infestations can cause significant crop losses, threatening food security and economic stability worldwide. Traditional detection methods typically rely on manual inspection, which is time-consuming, labor-intensive, and prone to human error. With advancements in computer vision and deep learning, automated pest detection systems have gained attention for their potential to offer rapid and accurate identification of pest species in real-world agricultural environments (<xref ref-type="bibr" rid="B31">Rai and Sun, 2024</xref>). Developing robust pest detection models, however, remains challenging for several reasons. First, pest insects exhibit high intra-class variability (e.g., different developmental stages such as eggs, larvae, pupae, and adults) and low inter-class variability (similar appearances across species). This contrast often complicates accurate feature extraction and classification (<xref ref-type="bibr" rid="B6">Butera et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B29">Popescu et&#xa0;al., 2023</xref>). Second, constructing large-scale annotated datasets is difficult because gathering and labeling images for numerous pest species is resource-intensive and requires domain expertise. (<xref ref-type="bibr" rid="B18">Li, Y et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B44">Yang et&#xa0;al., 2022</xref>). In pest object detection, there are a huge number of pest types, and it is extremely costly or even impossible to directly detect all species. Collecting large-scale pest datasets is also highly challenging. In practice, pest management predominantly targets crops, with timely response to primary pests being essential (<xref ref-type="bibr" rid="B1">Ali et al., 2024</xref>). Rapidly collecting a small number of samples for these major pests can be more practical and cost-effective. Therefore, few-shot object detection (FSOD) has significant research value in pest management, as it enables effective detection of critical pests with minimal annotated data.</p>
<p>Few-shot learning (FSL) has emerged as a promising solution to address the problem of limited annotated data (<xref ref-type="bibr" rid="B18">Li, Y et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B20">Li X. et&#xa0;al., 2023</xref>). FSL aims to recognize new classes using only a few labeled examples by leveraging prior knowledge learned from other tasks or classes. In the context of pest detection, FSL can enable models to identify novel pest species with minimal labeled samples, which is highly valuable for practical agricultural applications. Despite the progress in FSL for image classification tasks, applying FSL to object detection, especially for small and densely packed pest insects, remains a significant challenge (<xref ref-type="bibr" rid="B15">Huang et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B28">P&#xf6;hler et&#xa0;al., 2023</xref>). Traditional object detection models like Faster R-CNN (<xref ref-type="bibr" rid="B32">Ren et&#xa0;al., 2016</xref>) struggle with small objects due to insufficient feature representation and the dominance of background information (<xref ref-type="bibr" rid="B36">Teng et&#xa0;al., 2022</xref>). Moreover, the high similarity between different pest species further complicates accurate detection and classification.</p>
<p>To overcome these challenges, we propose a novel FSOD framework specifically designed for pest insects, integrating feature aggregation and contrastive learning techniques. Our approach builds upon the Faster R-CNN architecture and introduces a Feature Aggregation Module (FAM) that leverages multi-scale features from both support and query images. By employing an attention mechanism, the model effectively fuses rich contextual information from the support set to enhance the representation of multi pest objects in the query images. Additionally, we incorporate SCL to improve the discriminative ability of the model. Contrastive learning has shown effectiveness in enhancing feature representations by pulling together samples of the same class and pushing apart samples of different classes (<xref ref-type="bibr" rid="B35">Sun et&#xa0;al., 2021</xref>). By integrating contrastive learning into the detection framework, we aim to increase intra-class compactness and inter-class variance, which is crucial for distinguishing between visually similar pest species.</p>
<p>Furthermore, we address the issue of class imbalance inherent in pest detection datasets by introducing a balancing mechanism in the loss function. We adopt the focal loss to focus the training on hard examples and underrepresented classes, thereby improving the model&#x2019;s robustness and accuracy (<xref ref-type="bibr" rid="B18">Li, Y et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B42">Wen et&#xa0;al., 2022</xref>).</p>
<sec id="s1_1">
<label>1.1</label>
<title>Key contributions include</title>
<sec id="s1_1_1">
<label>1.1.1</label>
<title>Feature aggregation module</title>
<p>We design a novel Feature Aggregation Module (FAM) that enhances the representation of multiple pest objects by aggregating multi-scale features from support and query images using an attention mechanism.</p>
</sec>
<sec id="s1_1_2">
<label>1.1.2</label>
<title>Supervised contrastive learning</title>
<p>We integrate SCL into the object detection framework to improve feature discrimination, promoting intra-class similarity and inter-class dissimilarity among pest species.</p>
</sec>
<sec id="s1_1_3">
<label>1.1.3</label>
<title>Balancing mechanism</title>
<p>We introduce a balancing mechanism in the loss function using focal loss to mitigate the impact of class imbalance in pest detection datasets.</p>
</sec>
<sec id="s1_1_4">
<label>1.1.4</label>
<title>Comprehensive evaluation</title>
<p>We conduct extensive experiments on benchmark pest detection datasets to validate the effectiveness of our proposed method, demonstrating significant improvements over baseline models.</p>
<p>The proposed method provides a practical solution for agricultural pest management by enabling accurate detection of critical pests with minimal annotated data (<xref ref-type="bibr" rid="B30">Ragu and Teo, 2023</xref>). Its ability to handle few-shot scenarios ensures timely responses to pest outbreaks, reducing reliance on pesticides and promoting sustainable practices.</p>
<p>The remainder of this paper is organized as follows: Section 2 reviews related work on pest detection, few-shot learning, and contrastive learning. Section 3 details our proposed methodology, including the Feature Aggregation Module (FAM), the SCL approach, and the multi-task loss function. Section 4 presents experimental setups and results, and Section 5 concludes with future directions for research.</p>
</sec>
</sec>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<sec id="s2_1">
<label>2.1</label>
<title>Pest detection in agriculture</title>
<p>The application of deep learning techniques in agriculture, particularly for pest detection, has gained momentum in recent years (<xref ref-type="bibr" rid="B29">Popescu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B26">Mahmood et&#xa0;al., 2023</xref>). Traditional pest detection methods often rely on manual scouting, which is inefficient and prone to human error (<xref ref-type="bibr" rid="B6">Butera et&#xa0;al., 2021</xref>). Deep learning-based approaches offer automated, accurate, and real-time detection capabilities, vital for integrated pest management systems.</p>
<p>Several studies have focused on object detection models tailored for pest insects. For instance, <xref ref-type="bibr" rid="B27">Pang et&#xa0;al. (2022)</xref> proposed an improved YOLOv4 algorithm for real-time pest detection in orchards, reportedly achieving high detection accuracy (mAP above 80%) with efficient processing speeds. Similarly, <xref ref-type="bibr" rid="B42">Wen et&#xa0;al. (2022)</xref> introduced Pest-YOLO to detect dense, tiny pests, attaining about 92% detection accuracy on large-scale datasets.</p>
<p>Despite these successes, both methods relied on substantial annotated data, which is often infeasible given the vast diversity of pest species and the complexity of field conditions (<xref ref-type="bibr" rid="B25">Liu et&#xa0;al., 2022</xref>). Moreover, many pests are small or densely clustered, challenging conventional detectors that struggle with small-object detection (<xref ref-type="bibr" rid="B36">Teng et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B46">Yang et&#xa0;al., 2024</xref>).</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Few-shot learning in agriculture</title>
<p>Few-shot learning (FSL) has emerged as a solution to data scarcity. In agriculture, FSL has been applied to tasks like plant disease recognition (<xref ref-type="bibr" rid="B18">Li, Y et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B44">Yang et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B4">Arg&#xfc;eso et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B8">Chen et al., 2021</xref>) and pest detection (<xref ref-type="bibr" rid="B20">Li X. et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B33">Rezaei et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B19">Li Y. et al., 2023</xref>). Li and Chao (<xref ref-type="bibr" rid="B18">Li and Chao, 2021</xref>) proposed a semi-supervised few-shot learning approach for plant disease recognition, leveraging unlabeled data to improve classification when labeled samples are limited. <xref ref-type="bibr" rid="B44">Yang et&#xa0;al. (2022)</xref>, <xref ref-type="bibr" rid="B7">Cao et al. (2023)</xref>, <xref ref-type="bibr" rid="B21">Liang et al. (2021)</xref>, <xref ref-type="bibr" rid="B22">Lin et al. (2024)</xref> and <xref ref-type="bibr" rid="B23">Lin et al. (2022a)</xref> highlighted the role of FSL in smart agriculture, noting its effectiveness for rapid adaptation to new conditions or pest species. In pest detection, few-shot learning enables models to generalize to new pests with only a handful of labeled samples, a crucial capability given the difficulty of obtaining comprehensive data for every pest species. <xref ref-type="bibr" rid="B20">Li X. et&#xa0;al. (2023)</xref> introduced a few-shot crop pest detection method using object pyramids, reporting a notable increase in mAP under low-data conditions. <xref ref-type="bibr" rid="B33">Rezaei et&#xa0;al. (2024)</xref>, <xref ref-type="bibr" rid="B10">Egusquiza et al. (2022)</xref> and <xref ref-type="bibr" rid="B49">Zhou et al. (2023)</xref> demonstrated that even modest improvements in few-shot scenarios significantly impacted real-world applications, reinforcing the practicality of FSL in pest management (<xref ref-type="bibr" rid="B11">Gao et al., 2024</xref>). Nonetheless, effectively transferring FSL methods from classification to object detection remains challenging (<xref ref-type="bibr" rid="B39">Wang C. et al., 2023</xref>; <xref ref-type="bibr" rid="B41">Wang et al., 2021</xref>), especially under severe data constraints and small-object settings.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Contrastive learning and feature representation</title>
<p>Contrastive learning has gained attention for learning discriminative feature representations by contrasting positive and negative sample pairs (<xref ref-type="bibr" rid="B35">Sun et&#xa0;al., 2021</xref>). In FSOD, contrastive learning helps models differentiate classes with limited samples by enlarging inter-class separation within the feature space. <xref ref-type="bibr" rid="B35">Sun et&#xa0;al. (2021)</xref>. proposed FSCE, which encodes proposals using contrastive learning to enhance detection performance in few-shot settings, reportedly improving mAP on benchmark datasets by up to 3&#x2013;5 percentage points. In agricultural applications, contrastive learning has also been employed to improve classification. <xref ref-type="bibr" rid="B34">Song et&#xa0;al. (2023)</xref> and <xref ref-type="bibr" rid="B48">Zhong et&#xa0;al. (2020)</xref> used an attention-based generative adversarial network with few-shot learning to boost feature representation for maize disease detection, achieving higher accuracy scores compared to baseline CNN models. These results suggest that contrastive learning can likewise benefit the detection of various agricultural pests, particularly when data are limited or imbalanced.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Feature aggregation techniques</title>
<p>Feature aggregation combines features from different layers or sources to improve detection performance. For small-object detection, multi-scale feature fusion can be critical (<xref ref-type="bibr" rid="B17">Kong et al., 2024</xref>; <xref ref-type="bibr" rid="B24">Lin et al., 2022b</xref>). <xref ref-type="bibr" rid="B36">Teng et&#xa0;al. (2022)</xref> developed MSR-RCNN, integrating multi-scale super-resolution enhancements, increasing detection accuracy for small pest objects by around 4% in mAP. <xref ref-type="bibr" rid="B13">Han J. et&#xa0;al. (2023)</xref> presented a FSOD method using variational feature aggregation, demonstrating substantial improvements under limited-data conditions.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Addressing class imbalance</title>
<p>Class imbalance is pervasive in pest detection, where certain dominant pest species overshadow minority ones (<xref ref-type="bibr" rid="B42">Wen et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B25">Liu et&#xa0;al., 2022</xref>). Focal loss has proven effective in re-weighting hard examples and mitigating bias toward majority classes (<xref ref-type="bibr" rid="B18">Li, Y et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B42">Wen et&#xa0;al., 2022</xref>). Anwar and Masood (<xref ref-type="bibr" rid="B3">Anwar and Masood, 2023</xref>) also emphasized the importance of addressing imbalance, demonstrating a 5-8% improvement in detection accuracy by incorporating focal loss and augmenting minority classes.</p>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Advances in few-shot object detection</title>
<p>Recent surveys by <xref ref-type="bibr" rid="B15">Huang et&#xa0;al. (2023)</xref> and <xref ref-type="bibr" rid="B28">P&#xf6;hler et&#xa0;al. (2023)</xref> extensively review FSOD methods, including meta-learning, transfer learning, and metric learning techniques. The Segment Anything Model (SAM) (<xref ref-type="bibr" rid="B47">Zhang et&#xa0;al., 2023</xref>) represents a significant advancement in vision models, generalizing to new tasks with minimal data. While SAM primarily targets segmentation, it could be adapted for object detection under few-shot scenarios. Further, <xref ref-type="bibr" rid="B14">Han et&#xa0;al. (2023)</xref> extended SAM to open-vocabulary learning, enabling zero-shot generalization to unseen classes. These advancements suggest promising directions for applying cutting-edge few-shot methods to pest detection tasks.</p>
<p>Despite these advancements, several challenges persist in pest detection. First, multi-object detection remains problematic, as many models fail to handle multiple, densely packed pest insects due to insufficient feature representation (<xref ref-type="bibr" rid="B36">Teng et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B46">Yang et&#xa0;al., 2024</xref>). Second, labeled data scarcity restricts models from generalizing to novel pests, especially when each species demands expert-labeled samples (<xref ref-type="bibr" rid="B18">Li, Y et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B44">Yang et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B25">Liu et&#xa0;al., 2022</xref>). Third, visual similarity among pests complicates accurate feature discrimination (<xref ref-type="bibr" rid="B6">Butera et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B29">Popescu et&#xa0;al., 2023</xref>). Finally, class imbalance skews detection results, disadvantaging minority species (<xref ref-type="bibr" rid="B42">Wen et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B25">Liu et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B38">Wang X. et al., 2023</xref>). Our proposed method addresses these issues by incorporating feature aggregation to improve multi-object representation, SCL to enhance feature discrimination for visually similar pests, and a balancing mechanism to correct dataset imbalance.</p>
<p>In doing so, we aim to advance the state of pest detection by boosting accuracy for small, minority-class targets, reinforcing the practicality of few-shot techniques in agricultural domains.</p>
</sec>
<sec id="s2_7">
<label>2.7</label>
<title>Comparison of existing pest recognition methods</title>
<p>As shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, the comparison table summarizes various pest recognition methods, highlighting the differences in tasks, architectures, and small-shot learning capabilities. Previous research on pest identification and detection largely relied on CNN-based architectures, including YOLO and Faster R-CNN, which offered effective solutions for recognizing and localizing pests but struggled with challenges like detecting tiny pests, distinguishing visually similar species, and addressing class imbalance. Although recent works introduced improvements, such as multi-scale feature fusion, super-resolution sampling, and focal-loss-based imbalance handling, they generally addressed these issues separately rather than in a unified framework. Few-shot methods, while beneficial for scenarios with limited training data, were often limited to classification tasks without explicit handling of small pests or class imbalance. In contrast, the method proposed in this paper innovatively integrates multi-scale feature aggregation, supervised contrastive learning, and focal loss within a unified Faster R-CNN framework. Feature aggregation significantly improves multi-object detection by fusing multi-scale features, while supervised contrastive learning enhances discriminative capabilities by effectively differentiating similar pest species even from minimal examples. Additionally, focal loss addresses class imbalance by prioritizing minority-class and challenging samples during training. Consequently, this comprehensive approach robustly tackles key limitations of existing methods, achieving superior detection accuracy and better generalization to novel and rare pest species, demonstrating significant practical value for real-world agricultural applications under limited labeled data conditions.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Comparison of existing pest recognition methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Function</th>
<th valign="middle" align="center">Architecture</th>
<th valign="middle" align="center">Contrastive learning</th>
<th valign="middle" align="center">Multi-target and multi-scale</th>
<th valign="middle" align="center">Class <break/>imbalance <break/>handling</th>
<th valign="middle" align="center">Representative papers</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="4" align="center">Recognition (Classification)</td>
<td valign="middle" align="center">Deep CNNs</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B29">Popescu et&#xa0;al. (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">Deep CNNs with ensemble-based mode</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B3">Anwar and Masood (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">CNN+Transformer</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B2">An et&#xa0;al. (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">Transformer+super resolution sampling technique</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B5">Bai et&#xa0;al. (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" rowspan="2" align="center">Recognition (Classification)<break/>Few-Shot</td>
<td valign="middle" align="center">Transformers</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B37">uthalapati and Tunga (2021)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">a multi-layer feature<break/>fusion (FMLF) method</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">Yes</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B12">Gomes et&#xa0;al. (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="center">Object Detection (Classification/position)</td>
<td valign="middle" align="center">Deep CNNs</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B6">Butera et&#xa0;al. (2021)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">Multi-target and multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B27">Pang et&#xa0;al. (2022)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">Faster R-CNN</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B39">Wang C. et&#xa0;al. (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">Pest-YOLO</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">Yes</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B42">Wen et&#xa0;al. (2022)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">multi-scale super-resolution RCNN</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B36">Teng et&#xa0;al. (2022)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">SRNet-YOLO</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B46">Yang et&#xa0;al. (2024)</xref>
</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="center">Object Detection (Classification/position)<break/>Few-Shot</td>
<td valign="middle" align="center">Faster R-CNN</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B20">Li X. et&#xa0;al. (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">Faster R-CNN</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B45">Yang et&#xa0;al. (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">Faster R-CNN</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">multi-scale</td>
<td valign="middle" align="center">No</td>
<td valign="middle" align="center">
<xref ref-type="bibr" rid="B38">Wang X. et&#xa0;al. (2023)</xref>
</td>
</tr>
<tr>
<td valign="middle" align="center">Faster R-CNN</td>
<td valign="middle" align="center">Yes</td>
<td valign="middle" align="center">Multi-target and multi-scale</td>
<td valign="middle" align="center">Yes</td>
<td valign="middle" align="center">Proposed Method</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Proposed methodology</title>
<p>Our research presents an improved model based on the Faster R-CNN framework, aiming to enhance the feature representation capability of small-sample targets and improve object detection performance. Traditional Faster R-CNN frameworks face performance bottlenecks when handling small samples and multi object detection, primarily due to limitations in feature extraction layers and insufficient representation of small object features. To address these issues, we introduce multi-scale feature extraction for the support set and query set, expanding the capacity of feature extraction.</p>
<p>After feature extraction, the model inputs the features of the support set and query set into the Feature Aggregation Module (FAM). This module employs an attention mechanism for relational modeling, calculating the correlation between the support set and query set to construct aggregated features for multi-scale and multi objects. This feature aggregation method effectively utilizes the rich feature information from the support set, enhancing the feature representation capability of the query set, especially for detecting small-sample targets.</p>
<p>To further improve the model&#x2019;s discriminative ability, we incorporate SCL. By performing contrastive learning mapping and normalization on features, we enhance intra-class similarity and inter-class dissimilarity, promoting the clustering of similar samples and the separation of dissimilar samples in the feature space, thereby alleviating misclassification issues. However, SCL may suffer from sample imbalance problems, where insufficient samples of minority classes may cause the model to bias toward majority classes. To resolve this, we introduce an imbalance correction mechanism, adopting Focal Loss to optimize the loss function, assigning higher weights to hard-to-classify samples, and balancing the influence of each class.</p>
<p>Finally, we adopt a multi-task learning approach to jointly optimize four tasks: localization, classification, feature aggregation, and SCL. By integrating these components into the model, we achieve efficient detection of multi-sample targets, enhancing the model&#x2019;s feature representation capability and classification accuracy.</p>
<sec id="s3_1">
<label>3.1</label>
<title>Framework overview</title>
<p>As shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, our proposed model is built on the Faster R-CNN framework and is enhanced to effectively represent the features of multi-sample objects. The model architecture comprises several key components: first, multi-scale feature extraction, which integrates a Feature Pyramid Network (FPN) into the backbone network to capture rich information across various scales. Second, the Feature Aggregation Module (FAM), an attention-based component, aggregates features from both the support set and query set, enhancing the representation of multi-scale objects. Third, the SCL module improves the discriminative ability of the feature space by maximizing intra-class similarity and inter-class differences. Fourth, an imbalance correction mechanism incorporates focal loss into the loss function to address sample imbalance, ensuring the model focuses more on minority classes and challenging examples. Finally, the multi-task learning optimization jointly optimizes localization, classification, feature aggregation, and contrastive learning tasks through a comprehensive loss function. This integration enables the model to exploit contextual and class-specific information from the support set, significantly improving detection performance for both multi-object and few-shot objects. While Faster R-CNN is known to struggle with small-object detection due to insufficient feature representation, it was chosen for its robust two-stage detection process, which ensures precise localization and classification. The integration of FAM and SCL addresses its limitations by enhancing feature representation and improving discrimination for small objects. Comparative results show that the proposed enhancements improve mAP for small objects compared to the unmodified Faster R-CNN.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Overall model architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g001.tif"/>
</fig>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Feature aggregation module</title>
<p>The core objective of the Feature Aggregation Module (FAM) is to utilize the rich feature information from the support set to enhance the feature representation capability of the query set, especially for detecting multi and multi-scale objects. Traditional feature extraction methods have limited ability to represent multi object features, whereas the support set provides additional context and class information to compensate for this deficiency.</p>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>Multi-scale feature extraction</title>
<p>We integrate a Feature Pyramid Network (FPN) into the backbone network to extract features from different scales. Specifically, we obtain feature maps from multiple levels (C2, C3, C4, C5) of the backbone network (e.g., ResNet) and generate multi-scale feature maps <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>6</mml:mn>
</mml:msub>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> through 1&#xd7;1 and 3&#xd7;3 convolution operations. This multi-scale feature extraction ensures the model&#x2019;s sensitivity to targets of various sizes.</p>
<p>For each Region of Interest (RoI) in the support set and query set, we perform RoI Align operations on these multi-scale feature maps to obtain fixed-size feature representations (e.g., 7&#xd7;7). These feature representations preserve spatial information and contextual relationships, providing rich features for subsequent feature aggregation.</p>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>Structure of the feature aggregation module</title>
<p>The Feature Aggregation Module (FAM) consists of the following components: Feature Mapping, maps the features of the support set and query set into query (Q), key (K), and value (V) spaces, as shown in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>. Attention Mechanism, calculates the similarity between queries and keys to obtain the attention weight matrix. Feature Fusion, uses attention weights to perform weighted summation of values, achieving feature aggregation.</p>
<p>Implementation Details:</p>
<sec id="s3_2_2_1">
<label>3.2.2.1</label>
<title>Mapping features to query, key, and value spaces</title>
<p>First, we map the features of the support set and query set into low-dimensional spaces through linear transformations as shown in <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>K</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>K</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>V</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>V</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the feature representations of the query set and support set, respectively, and <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>Q</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>K</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msup>
<mml:mi>W</mml:mi>
<mml:mi>V</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are learnable parameter matrices.</p>
</sec>
<sec id="s3_2_2_2">
<label>3.2.2.2</label>
<title>Calculating attention weights</title>
<p>Using the dot product between queries and keys, we calculate the similarity scores and normalize them through the softmax function as shown in <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtext>A</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mtext>softmax,</mml:mtext>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>QK</mml:mtext>
</mml:mrow>
<mml:mo>&#x22a4;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mtext>d</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mtext>d</mml:mtext>
<mml:mtext>k</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the dimension of the key vectors, used to scale the dot product to prevent excessively large values.</p>
</sec>
<sec id="s3_2_2_3">
<label>3.2.2.3</label>
<title>Feature aggregation</title>
<p>Using the attention weight matrix A to perform weighted summation of the values V, the aggregated feature representation is formulated as <xref ref-type="disp-formula" rid="eq3">Equation 3</xref>.</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Then, we fuse the aggregated features with the original query set features to obtain the enhanced feature representation, defined in <xref ref-type="disp-formula" rid="eq4">Equation 4</xref>.</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>q</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mtext>&#x3b1;</mml:mtext>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>g</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im8">
<mml:mtext>&#x3b1;</mml:mtext>
</mml:math>
</inline-formula> is a learnable scaling factor that controls the influence of the aggregated features on the original features.</p>
</sec>
</sec>
<sec id="s3_2_3">
<label>3.2.3</label>
<title>Aggregated features for multi-scale and multi objects</title>
<p>Through the feature aggregation process described above, the feature representation of the query set is enhanced in several ways. Multi-scale information fusion leverages features from the support set&#x2019;s multi-scale feature maps, providing rich scale information that aids in detecting targets of various sizes. Multi object feature enhancement is achieved by supplementing high-level feature maps with low-level features from the support set, which preserves details that are often lost for multi objects. Additionally, the support set&#x2019;s features provide valuable contextual information, helping the model understand the relationship between the target and its background. The module offers several advantages: it improves feature representation by effectively utilizing the rich information from the support set, enhances flexibility and scalability through an attention-based relational modeling approach that adaptively adjusts the influence of the support set on the query set, and allows for easy integration into existing object detection frameworks with minimal computational overhead.</p>
</sec>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Supervised contrastive learning module</title>
<p>In object detection tasks, the model&#x2019;s discriminative ability is crucial for detection accuracy. However, due to the dispersion of intra-class features and the overlap of inter-class features, the model may experience misclassification issues. To address this problem, we introduce SCL, aiming to optimize the feature space so that features of the same class are closer together, while features of different classes are farther apart.</p>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Contrastive learning feature mapping and normalization</title>
<p>We apply a projection head <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mtext>Head</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to the enhanced features <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> output from the Feature Aggregation Module (FAM) to map them into the contrastive learning feature space, as expressed by <xref ref-type="disp-formula" rid="eq5">Equation 5</xref>.</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mtext>z</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>Normalize</mml:mtext>
<mml:mo>(</mml:mo>
<mml:mtext>Head</mml:mtext>
<mml:mo>(</mml:mo>
<mml:msub>
<mml:mtext>F</mml:mtext>
<mml:mrow>
<mml:mtext>enhanced</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>i</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where Normalize (&#xb7;) denotes <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mtext>L</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> normalization to ensure the feature vectors lie on a unit hypersphere.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Contrastive learning feature mapping and normalization</title>
<p>In SCL, label information is used to construct positive and negative sample pairs. Positive samples consist of a query sample <italic>i</italic> and support samples <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:mtext>p</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mtext>P</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>i</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> that belong to the same class. Negative samples consist of the query sample <italic>i</italic> and support samples <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:mtext>a</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>i</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> from different classes. Here, <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> represents the set of samples in the same class as sample <italic>i</italic>, while <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> represents the set of all samples except sample <italic>i</italic>.</p>
</sec>
<sec id="s3_3_3">
<label>3.3.3</label>
<title>Supervised contrastive loss function</title>
<p>We adopt the supervised contrastive loss function to optimize the feature representation, which is formulated as <xref ref-type="disp-formula" rid="eq6">Equation 6</xref>.</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mtext>L</mml:mtext>
<mml:mrow>
<mml:mtext>SCL</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext>i</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mtext>I</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mtext>P</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>i</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext>p</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mtext>P</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>i</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>z</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mtext>z</mml:mtext>
<mml:mtext>p</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mtext>&#x3c4;</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mtext>a</mml:mtext>
<mml:mo>&#x2208;</mml:mo>
<mml:mtext>A</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mtext>i</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mi>exp</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>z</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mtext>z</mml:mtext>
<mml:mtext>a</mml:mtext>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mtext>&#x3c4;</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im17">
<mml:mtext>&#x3c4;</mml:mtext>
</mml:math>
</inline-formula> is the temperature parameter controlling the smoothness of the distribution.</p>
<p>By minimizing the supervised contrastive loss, the model is guided to achieve intra-class compactness, where features of the same class are closer together, enhancing similarity within each class. It also promotes inter-class separation, pushing features of different classes farther apart and increasing dissimilarity between classes. This optimization helps the model classify more accurately and reduces misclassification.</p>
</sec>
<sec id="s3_3_4">
<label>3.3.4</label>
<title>Correction for sample imbalance</title>
<p>SCL may be affected by sample imbalance, where minority classes have insufficient samples, causing the model to bias toward majority classes. To address this, we introduce an imbalance correction mechanism.</p>
<p>Specifically, we incorporate class weights <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msub>
<mml:mtext>w</mml:mtext>
<mml:mrow>
<mml:msub>
<mml:mtext>y</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> into the supervised contrastive loss, adjusting according to the number of samples in each class, as defined in <xref ref-type="disp-formula" rid="eq7">Equation 7</xref>.</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:msub>
<mml:mtext>N</mml:mtext>
<mml:mrow>
<mml:msub>
<mml:mtext>y</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of samples in class <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msub>
<mml:mtext>y</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The loss function <xref ref-type="disp-formula" rid="eq6">Equation 6</xref> becomes <xref ref-type="disp-formula" rid="eq8">Equation 8</xref>.</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>SCL</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mtext>&#x3c4;</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:msub>
<mml:mi>exp</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mtext>&#x3c4;</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>This adjustment prioritizes minority class samples in the loss function, prompting the model to focus more on learning these classes.</p>
<p>The introduction of SCL will enhance the discriminability of the feature space, reduce misclassification, and thus improve classification accuracy; at the same time, through the imbalance correction mechanism, the model can learn the minority classes more fully and adapt to imbalanced data; finally, better feature representation helps the model perform better on unknown data and enhances generalization capabilities.</p>
</sec>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Overall loss function design</title>
<sec id="s3_4_1">
<label>3.4.1</label>
<title>Construction of the multi-task loss function</title>
<p>To jointly optimize the model&#x2019;s components, we design a comprehensive multi-task loss function that includes localization loss, classification loss, feature aggregation loss, and supervised contrastive loss, which is defined in <xref ref-type="disp-formula" rid="eq9">Equation 9</xref>.</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>total</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>cls</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>reg</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>&#x3bb;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>agg</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>&#x3bb;</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>SCL</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Among them, <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>cls</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the classification loss, which measures the model&#x2019;s prediction accuracy of the target class, <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>reg</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the regression loss, which measures the model&#x2019;s positioning accuracy of the target bounding box, <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>agg</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the loss of the Feature Aggregation Module (FAM), which may include the regularization term of the attention mechanism, <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>SCL</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the supervised contrast loss, which enhances the discriminability of feature representation, <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3bb;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mtext>&#xa0;and&#xa0;&#x3bb;</mml:mtext>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are trade-off coefficients that adjust the impact of each loss term.</p>
</sec>
<sec id="s3_4_2">
<label>3.4.2</label>
<title>Design of the classification loss</title>
<p>We employ Focal Loss for the classification loss <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:msub>
<mml:mtext>L</mml:mtext>
<mml:mrow>
<mml:mtext>cls</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> to address sample imbalance, especially in scenarios with imbalanced positive and negative samples. The Focal Loss is defined as <xref ref-type="disp-formula" rid="eq10">Equation 10</xref>.</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>cls</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mtext>&#x3b1;</mml:mtext>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mtext>&#x3b3;</mml:mtext>
</mml:msup>
<mml:mi>log</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Among them, <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the model&#x2019;s predicted probability of the true class, <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3b1;</mml:mtext>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the class weight, which balances the impact of the number of samples in different classes, and <inline-formula>
<mml:math display="inline" id="im29">
<mml:mrow>
<mml:mtext>&#x3b3;&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> is the adjustment factor, which reduces the loss contribution of easy samples and focuses on hard samples. Through Focal Loss, we can reduce the impact of a large number of easy negative samples on the loss, so that the model can pay more attention to hard positive samples.</p>
</sec>
<sec id="s3_4_3">
<label>3.4.3</label>
<title>Design of the regression loss</title>
<p>For the regression loss <inline-formula>
<mml:math display="inline" id="im30">
<mml:mrow>
<mml:msub>
<mml:mtext>L</mml:mtext>
<mml:mrow>
<mml:mtext>reg</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, measuring bounding box localization accuracy, we use the Smooth <inline-formula>
<mml:math display="inline" id="im31">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> Loss as expressed in <xref ref-type="disp-formula" rid="eq11">Equation 11</xref>.</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>reg</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo>{</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mtext>smooth</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:msub>
<mml:mtext>t</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted bounding box parameter, <inline-formula>
<mml:math display="inline" id="im33">
<mml:mrow>
<mml:msub>
<mml:mtext>v</mml:mtext>
<mml:mtext>i</mml:mtext>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the truth bounding box parameter, <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> are location information of the box and smooth<sub>L1</sub> (&#xb7;) is the Smooth loss function.</p>
</sec>
<sec id="s3_4_4">
<label>3.4.4</label>
<title>Modeling of the feature aggregation loss</title>
<p>To ensure effective utilization of support set information and prevent overfitting or redundancy due to the attention mechanism, we introduce the feature aggregation loss <inline-formula>
<mml:math display="inline" id="im36">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>agg</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, consisting of attention regularization and sparsity constraint.</p>
<sec id="s3_4_4_1">
<label>3.4.4.1</label>
<title>Attention regularization</title>
<p>We use Attention Entropy as a regularization term to prevent attention weights from over-concentrating on a few support samples, encouraging comprehensive utilization of support set information.</p>
<p>Attention Weight Matrix, for query set sample <italic>i</italic> and support set sample <italic>j</italic>, the attention weight <inline-formula>
<mml:math display="inline" id="im37">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is computed as <xref ref-type="disp-formula" rid="eq12">Equation 12</xref>:</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mi>exp</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where the similarity score <inline-formula>
<mml:math display="inline" id="im38">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is: <inline-formula>
<mml:math display="inline" id="im39">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>q</mml:mi>
</mml:mstyle>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>i</mml:mi>
</mml:mstyle>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>k</mml:mi>
</mml:mstyle>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>j</mml:mi>
</mml:mstyle>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im40">
<mml:mrow>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>q</mml:mi>
</mml:mstyle>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>i</mml:mi>
</mml:mstyle>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the query vector of <italic>i</italic> query sample, <inline-formula>
<mml:math display="inline" id="im41">
<mml:mrow>
<mml:msub>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>k</mml:mi>
</mml:mstyle>
<mml:mstyle mathvariant="bold" mathsize="normal">
<mml:mi>j</mml:mi>
</mml:mstyle>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the key vector supporting sample <italic>j</italic>, and <inline-formula>
<mml:math display="inline" id="im42">
<mml:mrow>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the dimension of the key vector.</p>
<sec id="s3_4_4_1_1">
<label>3.4.4.1.1</label>
<title>Attention entropy regularization term</title>
<p>The attention entropy regularization term is formulated as <xref ref-type="disp-formula" rid="eq13">Equation 13</xref>.</p>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>attn_reg</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mo>(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>log</mml:mi>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im43">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the numbers of query and support samples.</p>
<p>By maximizing the attention entropy (i.e., minimizing the negative attention entropy), the attention weights are encouraged to be more evenly distributed over the support set, preventing over-reliance on a small number of support samples.</p>
</sec>
</sec>
<sec id="s3_4_4_2">
<label>3.4.4.2</label>
<title>Sparsity constraint</title>
<p>To encourage sparsity in attention weights, focusing on the most relevant support samples and enhancing discriminative power, we impose a sparsity constraint on the unnormalized similarity scores <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, Sparsity Regularization Term is defined in <xref ref-type="disp-formula" rid="eq14">Equation 14</xref>.</p>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>sparse</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>By summing the absolute values of the similarity scores, the model is encouraged to generate a sparser similarity matrix, making the attention weights more inclined to a small number of important support samples.</p>
<sec id="s3_4_4_2_1">
<label>3.4.4.2.1</label>
<title>Complete feature aggregation loss function</title>
<p>Combining the attention regularization <xref ref-type="disp-formula" rid="eq13">Equation 13</xref> and sparsity constraint <xref ref-type="disp-formula" rid="eq14">Equation 14</xref>, the feature aggregation loss is computed by <xref ref-type="disp-formula" rid="eq15">Equation 15</xref>.</p>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>agg</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>log</mml:mi>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>)</mml:mo>
<mml:mo>+</mml:mo>
<mml:mo>(</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>q</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>|</mml:mo>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>These regularization terms help the model better utilize the information of the support set and prevent the attention weights from being over-concentrated or over-dispersed, thereby improving the effect of feature aggregation and improving the detection performance of the model.</p>
</sec>
<sec id="s3_4_4_2_2">
<label>3.4.4.2.2</label>
<title>The choice of the balance coefficients</title>
<p>
<inline-formula>
<mml:math display="inline" id="im46">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3bb;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3bb;</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> has an important impact on the performance of the model. Usually, we can adjust the values of these coefficients through experimental verification to achieve the best performance. In general, the values of <inline-formula>
<mml:math display="inline" id="im48">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3bb;</mml:mtext>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3bb;</mml:mtext>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> can be set to 1 or scaled according to the relative size of the loss terms.</p>
<p>By jointly optimizing the above loss functions, our model can simultaneously achieve the optimization goals of classification accuracy, positioning accuracy, feature representation, and imbalance during training. That is, through Focal Loss and SCL, the model more accurately predicts the target class to improve classification accuracy. By optimizing regression loss, the model can more accurately locate the target boundary to improve positioning accuracy. Through feature aggregation and SCL, the model&#x2019;s feature expression ability is improved, thereby enhancing feature representation. The weight mechanism introduced in the loss function enables the model to pay more attention to minority classes and hard samples to adapt to unbalanced data.</p>
</sec>
</sec>
</sec>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Training strategies</title>
<sec id="s3_5_1">
<label>3.5.1</label>
<title>Multi-task joint training</title>
<p>We employed a multi-task learning approach to facilitate collaborative optimization among various model components. In each training iteration, localization loss, classification loss, feature aggregation loss, and supervised contrastive loss were computed. These losses were then combined into a single cumulative loss <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mtext>total</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Backpropagation and parameter updates were performed based on this total loss, ensuring joint optimization of all components. This approach promotes better feature representation, faster convergence, and improved generalization by encouraging mutual information sharing among tasks.</p>
</sec>
<sec id="s3_5_2">
<label>3.5.2</label>
<title>Learning rate and optimizer</title>
<p>To stabilize training and prevent initial oscillations, we adopted a piecewise or cosine annealing learning rate decay schedule. This strategy lowers the learning rate in a controlled manner, allowing the model to converge steadily. For the optimizer, we used Stochastic Gradient Descent (SGD) with momentum to accelerate convergence and smooth out gradients. The momentum factor was tuned on the validation set to achieve the best balance between convergence speed and stability.</p>
</sec>
<sec id="s3_5_3">
<label>3.5.3</label>
<title>Weight initialization</title>
<p>To expedite convergence and leverage prior knowledge, we initialized model parameters using ImageNet-pretrained backbone weights. Newly added modules, such as the Feature Aggregation Module (FAM) and the projection head for contrastive learning, were initialized using Kaiming initialization. This approach ensures that important structural components inherit robust feature representations while newly introduced parameters adapt rapidly.</p>
</sec>
<sec id="s3_5_4">
<label>3.5.4</label>
<title>Regularization and overfitting prevention</title>
<p>To mitigate overfitting, we applied L2 regularization (weight decay) in the optimizer. Additionally, dropout was introduced in fully connected layers and within the FAM to stochastically deactivate a fraction of neurons during training. This not only prevents the model from over-relying on specific neurons but also improves its capacity to generalize to unseen data.</p>
</sec>
<sec id="s3_5_5">
<label>3.5.5</label>
<title>Data augmentation</title>
<p>We employed a variety of image augmentation techniques, including random cropping, rotation, flipping, and color jittering, to increase data diversity and reduce overfitting. For class imbalance issues&#x2014;especially in few-shot scenarios&#x2014;we performed sample balancing by oversampling minority classes or undersampling majority classes, aiming to achieve a more balanced and representative training set. This augmentation and balancing strategy is particularly critical in 3-shot, 5-shot, and 10-shot experiments, where the training samples are limited.</p>
</sec>
<sec id="s3_5_6">
<label>3.5.6</label>
<title>Training process monitoring</title>
<p>We continuously tracked training progress by observing the loss curves (localization, classification, feature aggregation, and supervised contrastive) to ensure stable convergence. Model performance was periodically evaluated on a validation set using metrics such as mean average precision (mAP) or recall. If the performance began to plateau or degrade, we adjusted hyperparameters&#x2014;including learning rate, momentum, and regularization factors&#x2014;accordingly.</p>
</sec>
<sec id="s3_5_7">
<label>3.5.7</label>
<title>Hyperparameter adjustment</title>
<p>In our loss function, the coefficients <inline-formula>
<mml:math display="inline" id="im51">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im52">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> determine the relative importance of each sub-loss. We fine-tuned these values based on validation performance, ensuring that no single loss term dominated the training.</p>
<p>Of particular importance is the temperature parameter <inline-formula>
<mml:math display="inline" id="im53">
<mml:mtext>&#x3c4;</mml:mtext>
</mml:math>
</inline-formula> in the supervised contrastive loss, which controls the smoothness of the probability distribution when computing similarities among samples. Proper tuning of \(\tau\) helps stabilize contrastive learning by balancing the separation between positive and negative pairs. After addressing reviewer concerns, we corrected the temperature parameter usage by referencing the optimal settings reported in the official FSCE(Sun, B et&#xa0;al., 2021) experiment. Few-Shot Settings: For 3-shot, 5-shot, and 10-shot training, the positive sample IoU thresholds were set to 0.6, 0.7, and 0.8, respectively. Temperature Coefficients <inline-formula>
<mml:math display="inline" id="im54">
<mml:mtext>&#x3c4;</mml:mtext>
</mml:math>
</inline-formula>: Consistently set to 0.2 for 3-shot, 5-shot, and 10-shot. Aggregate Loss Weights <inline-formula>
<mml:math display="inline" id="im55">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>and Comparison Loss Weights <inline-formula>
<mml:math display="inline" id="im56">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: Set to 0.2, 0.5, and 0.5, respectively, in the 3-shot, 5-shot, and 10-shot scenarios.</p>
</sec>
<sec id="s3_5_8">
<label>3.5.8</label>
<title>Model saving and selection</title>
<p>To safeguard against unexpected interruptions, we regularly saved model checkpoints during training. Each checkpoint contained the model weights, optimizer state, and current learning rate. After completing training, we selected the best-performing checkpoint based on validation metrics for final testing and deployment. This ensures that the model used in downstream tasks represents the most robust and accurate version learned during training.</p>
<p>By implementing the above multi-task joint training strategy with detailed hyperparameter tuning, our model demonstrated stable and efficient training, fully harnessing the benefits of collaborative optimization. The experimental results (presented in Section 4) indicate marked improvements in both convergence speed and overall performance, corroborating the effectiveness of these methodologies. Additionally, fine-tuning in the two-stage Faster R-CNN architecture proved essential for adapting the model to specific datasets and tasks, yielding enhanced robustness and accuracy. This tailored approach ensures alignment with the unique characteristics of real-world applications, thereby solidifying the model&#x2019;s practical relevance.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiments</title>
<sec id="s4_1">
<label>4.1</label>
<title>Dataset, experimental configuration and parameter settings</title>
<p>This research introduces the PestDet dataset to support a few-shot pest detection method based on feature aggregation and SCL. PestDet, consisting of approximately 82,000 images, integrates data from the IP102 dataset (<xref ref-type="bibr" rid="B43">Wu et&#xa0;al., 2019</xref>), the IDADP dataset (<xref ref-type="bibr" rid="B9">Chen and Yuan, 2019</xref>), and additional images from the internet and production environments. It includes targets at individual, medium, collective, and mixed levels, covering various pest stages. The IP102 dataset, comprising over 75,000 images of 102 pests, served as the primary source, with 19,000 images containing detailed detection annotations. The IDADP dataset added 4,700 images of typical agricultural pests. Additional samples from tropical regions further enhanced dataset diversity.Dataset preprocessing included cleaning duplicate images using a pre-trained vision transformer (ViT), re-annotating different pest stages, and resolution equalization to balance image resolutions. Annotations were optimized by removing zero-area bounding boxes, duplicate boxes, and correcting incorrect labels. These steps improved dataset quality, ensuring effective training and better detection performance.</p>
<p>To construct the object detection dataset for this study, we leveraged the PestDet dataset, which was originally designed for classification and object detection tasks and includes 102 pest classes labeled from 0 to 101. <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> provides detailed statistics of the PestDet dataset, including the total number of images, the number of bounding boxes, and the number of single-bounding-box images for each pest class. However, the bounding box distribution in PestDet is highly imbalanced, with some classes having significantly more annotations than others, leading to a model bias toward classes with more bounding boxes during training. To address this issue and to focus on FSOD while considering computational constraints, we constructed a balanced subset, PestDet20, by selecting 20 pest classes from PestDet. These classes were chosen to represent pests commonly found in tropical and subtropical economic crops, characterized by individual diversity and complex backgrounds. The selection process, detailed in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, involved sorting all classes by bounding box count in descending order, excluding redundant or subset classes (e.g., those with large overlaps between larvae and adult forms of the same pest), and finally selecting the top 20 classes based on bounding box count. The selected classes are numbered {0, 3, 14, 15, 16, 21, 24, 25, 26, 37, 39, 48, 50, 66, 67, 70, 76, 95, 99, 101}, following the original PestDet numbering. Inspired by the 20-class structure of the PASCAL VOC dataset as outlined in the TFA standard, the PestDet20 dataset was constructed to provide a balanced and representative foundation for addressing the unique challenges of FSOD in pest management.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Overall class image information of training set and test set.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Metrics</th>
<th valign="middle" align="center">Number of images</th>
<th valign="middle" align="center">Number of total <break/>annotated boxes</th>
<th valign="middle" align="center">Number of <break/>images with <break/>unique annotated <break/>boxes</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">mean</td>
<td valign="middle" align="center">187</td>
<td valign="middle" align="center">217</td>
<td valign="middle" align="center">174</td>
</tr>
<tr>
<td valign="middle" align="center">std</td>
<td valign="middle" align="center">343</td>
<td valign="middle" align="center">362</td>
<td valign="middle" align="center">334</td>
</tr>
<tr>
<td valign="middle" align="center">min</td>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">5</td>
<td valign="middle" align="center">2</td>
</tr>
<tr>
<td valign="middle" align="center">25%</td>
<td valign="middle" align="center">44</td>
<td valign="middle" align="center">53</td>
<td valign="middle" align="center">31</td>
</tr>
<tr>
<td valign="middle" align="center">50%</td>
<td valign="middle" align="center">92.50</td>
<td valign="middle" align="center">120</td>
<td valign="middle" align="center">83</td>
</tr>
<tr>
<td valign="middle" align="center">75%</td>
<td valign="middle" align="center">183</td>
<td valign="middle" align="center">243</td>
<td valign="middle" align="center">175</td>
</tr>
<tr>
<td valign="middle" align="center">max</td>
<td valign="middle" align="center">2859</td>
<td valign="middle" align="center">2896</td>
<td valign="middle" align="center">2826</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To facilitate analysis and experimentation, a few-shot pest dataset, PestDet20, was created according to selected standards. Class statistics are summarized in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. The training set includes 5,076 images with 5,590 bounding boxes, while the testing set has 1,177 images and 1,292 bounding boxes, split by the typical 8:2 ratio. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> presents examples of the 20 pest classes studied. In the fine-tuning-based few-shot object detection task, the model training and testing process is divided into two stages: the base stage and the fine-tuning stage. The base stage consists of training and testing, where the training phase uses all samples of the base classes from the training set, and the testing phase uses all samples of the base classes from the testing set. Similarly, the fine-tuning stage also consists of training and testing. During the training phase of the fine-tuning stage, 3, 5, or 10 samples from both the base classes and the novel classes in the training set are used. For testing in the fine-tuning stage, all samples from both the base classes and the novel classes in the testing set are used.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Image information of 20 selected pests.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Class number</th>
<th valign="middle" align="center">Name</th>
<th valign="middle" align="center">Number of <break/>training set images</th>
<th valign="middle" align="center">Number of test set images</th>
<th valign="middle" align="center">Number of <break/>training set <break/>annotation boxes</th>
<th valign="middle" align="center">Number of test set annotation boxes</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">Rice leaf roller</td>
<td valign="middle" align="center">131</td>
<td valign="middle" align="center">34</td>
<td valign="middle" align="center">141</td>
<td valign="middle" align="center">38</td>
</tr>
<tr>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">Rice stem borer</td>
<td valign="middle" align="center">126</td>
<td valign="middle" align="center">33</td>
<td valign="middle" align="center">138</td>
<td valign="middle" align="center">34</td>
</tr>
<tr>
<td valign="middle" align="center">14</td>
<td valign="middle" align="center">Grub</td>
<td valign="middle" align="center">331</td>
<td valign="middle" align="center">80</td>
<td valign="middle" align="center">532</td>
<td valign="middle" align="center">108</td>
</tr>
<tr>
<td valign="middle" align="center">15</td>
<td valign="middle" align="center">Mole cricket</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
</tr>
<tr>
<td valign="middle" align="center">16</td>
<td valign="middle" align="center">Wireworm</td>
<td valign="middle" align="center">325</td>
<td valign="middle" align="center">80</td>
<td valign="middle" align="center">405</td>
<td valign="middle" align="center">104</td>
</tr>
<tr>
<td valign="middle" align="center">21</td>
<td valign="middle" align="center">Red spider</td>
<td valign="middle" align="center">125</td>
<td valign="middle" align="center">31</td>
<td valign="middle" align="center">128</td>
<td valign="middle" align="center">36</td>
</tr>
<tr>
<td valign="middle" align="center">24</td>
<td valign="middle" align="center">Aphid</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
</tr>
<tr>
<td valign="middle" align="center">25</td>
<td valign="middle" align="center">White-spotted flower beetle</td>
<td valign="middle" align="center">145</td>
<td valign="middle" align="center">40</td>
<td valign="middle" align="center">173</td>
<td valign="middle" align="center">46</td>
</tr>
<tr>
<td valign="middle" align="center">26</td>
<td valign="middle" align="center">Peach borer</td>
<td valign="middle" align="center">188</td>
<td valign="middle" align="center">45</td>
<td valign="middle" align="center">199</td>
<td valign="middle" align="center">47</td>
</tr>
<tr>
<td valign="middle" align="center">37</td>
<td valign="middle" align="center">Flea beetle</td>
<td valign="middle" align="center">253</td>
<td valign="middle" align="center">65</td>
<td valign="middle" align="center">285</td>
<td valign="middle" align="center">70</td>
</tr>
<tr>
<td valign="middle" align="center">39</td>
<td valign="middle" align="center">Beet armyworm</td>
<td valign="middle" align="center">317</td>
<td valign="middle" align="center">81</td>
<td valign="middle" align="center">322</td>
<td valign="middle" align="center">81</td>
</tr>
<tr>
<td valign="middle" align="center">48</td>
<td valign="middle" align="center">Acridoidea</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
</tr>
<tr>
<td valign="middle" align="center">50</td>
<td valign="middle" align="center">Blister beetle</td>
<td valign="middle" align="center">338</td>
<td valign="middle" align="center">85</td>
<td valign="middle" align="center">366</td>
<td valign="middle" align="center">94</td>
</tr>
<tr>
<td valign="middle" align="center">66</td>
<td valign="middle" align="center">Grape hawkmoth</td>
<td valign="middle" align="center">197</td>
<td valign="middle" align="center">53</td>
<td valign="middle" align="center">197</td>
<td valign="middle" align="center">53</td>
</tr>
<tr>
<td valign="middle" align="center">67</td>
<td valign="middle" align="center">Cicada</td>
<td valign="middle" align="center">253</td>
<td valign="middle" align="center">63</td>
<td valign="middle" align="center">253</td>
<td valign="middle" align="center">63</td>
</tr>
<tr>
<td valign="middle" align="center">70</td>
<td valign="middle" align="center">Lycophoridae</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
</tr>
<tr>
<td valign="middle" align="center">76</td>
<td valign="middle" align="center">Cotton scale</td>
<td valign="middle" align="center">121</td>
<td valign="middle" align="center">28</td>
<td valign="middle" align="center">216</td>
<td valign="middle" align="center">55</td>
</tr>
<tr>
<td valign="middle" align="center">95</td>
<td valign="middle" align="center">Brown-margined moth</td>
<td valign="middle" align="center">134</td>
<td valign="middle" align="center">33</td>
<td valign="middle" align="center">142</td>
<td valign="middle" align="center">36</td>
</tr>
<tr>
<td valign="middle" align="center">99</td>
<td valign="middle" align="center">Spine-chested longhorn beetle</td>
<td valign="middle" align="center">92</td>
<td valign="middle" align="center">26</td>
<td valign="middle" align="center">93</td>
<td valign="middle" align="center">27</td>
</tr>
<tr>
<td valign="middle" align="center">101</td>
<td valign="middle" align="center">Cicadidae</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
<td valign="middle" align="center">400</td>
<td valign="middle" align="center">80</td>
</tr>
<tr>
<td valign="middle" colspan="2" align="center">Total</td>
<td valign="middle" align="center">5076</td>
<td valign="middle" align="center">1177</td>
<td valign="middle" align="center">5590</td>
<td valign="middle" align="center">1292</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Examples of all classes of pests in the dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g002.tif"/>
</fig>
<p>Using the feature aggregation-based fine-tuning method from VFA (<xref ref-type="bibr" rid="B13">Han et&#xa0;al., 2023</xref>), the FSOD dataset was divided with a random shuffling strategy. The 20 selected pest classes {0, 3, 14, 15, 16, 21, 24, 25, 26, 37, 39, 48, 50, 66, 67, 70, 76, 95, 99, 101} were shuffled three times, creating distinct class arrays. In each shuffle, 15 classes served as base classes, while the remaining 5 were designated as novel classes, as shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Classification.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Split</th>
<th valign="middle" align="center">All classes</th>
<th valign="middle" align="center">Basic classes</th>
<th valign="middle" align="center">New classes</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">1</td>
<td valign="middle" align="left">{15, 76, 24, 39, 14, 67, 16, 95, 25, 3, 66, 0, 101, 99, 37, 70, 26, 50, 48, 21}</td>
<td valign="middle" align="left">{15, 76, 24, 39, 14, 67, 16, 95, 25, 3, 66, 0, 101, 99, 37}</td>
<td valign="middle" align="left">{70, 26, 50, 48, 21}</td>
</tr>
<tr>
<td valign="middle" align="center">2</td>
<td valign="middle" align="left">{101, 14, 48, 15, 3, 67, 39, 66, 76, 50, 95, 26, 37, 24, 0, 16, 21, 99, 70, 25}</td>
<td valign="middle" align="left">{101, 14, 48, 15, 3, 67, 39, 66, 76, 50, 95, 26, 37, 24, 0}</td>
<td valign="middle" align="left">{16, 21, 99, 70, 25}</td>
</tr>
<tr>
<td valign="middle" align="center">3</td>
<td valign="middle" align="left">{0, 48, 14, 99, 3, 21, 39, 66, 16, 37, 50, 26, 25, 70, 24, 67, 101, 76, 15, 95}</td>
<td valign="middle" align="left">{0, 48, 14, 99, 3, 21, 39, 66, 16, 37, 50, 26, 25, 70, 24}</td>
<td valign="middle" align="left">{67, 101, 76, 15, 95}</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The training set, used as the support set, and the testing set, used as the query set, evaluated the model&#x2019;s stability and robustness. Strong performance across subsets indicates model stability, while poor performance on certain subsets suggests sensitivity to specific classes or features. After dividing the base classes and novel classes, we trained and tested the model using 30 random seeds and obtained the average results to compare with methods that use random seeds. For the fine-tuning phase, we sampled images from each class to construct the training set, with the number of sampled images set to 3, 5, and 10, respectively. This approach ensures that the sample sizes of base classes and novel classes during the fine-tuning phase are balanced, thereby reducing the model&#x2019;s bias toward the base classes.</p>
<sec id="s4_1_1">
<label>4.1.1</label>
<title>Experimental configuration and parameter settings</title>
<p>Compared to few-shot classification and regular object detection tasks, FSOD faces more challenges. Its training dataset is mainly divided into two classes: base classes, with abundant annotated data, and novel classes, with limited annotated data. The main goal of FSOD is to significantly improve detection performance for novel classes while maintaining high detection accuracy for base class. FSOD effectively reduces the dependence of object detection models on large amounts of training data, solves the problem of imbalanced annotations in training data, and has significant practical value and a wide range of applications.</p>
<p>This study compared three classic FSOD algorithms: YOLO (<xref ref-type="bibr" rid="B16">Khanam and Hussain, 2024</xref>), TFA (<xref ref-type="bibr" rid="B40">Wang et&#xa0;al., 2020</xref>) VFAr43 (<xref ref-type="bibr" rid="B13">Han et&#xa0;al., 2023</xref>) and FSCE (<xref ref-type="bibr" rid="B35">Sun et&#xa0;al., 2021</xref>). Experiments were conducted on the Ubuntu operating system, using Python as the main development language, based on the PyTorch deep learning framework, with mmfewshot used for FSOD model training and testing. The hardware environment included two NVIDIA GeForce RTX 4090 GPUs with 24G VRAM each, an Intel(R) Xeon(R) CPU E5&#x2013;2680 v3, and 64G of memory.</p>
<p>In experimental hyperparameter settings, SGD was selected as the optimizer, with an initial learning rate of 0.02, a batch size of 4, and 18,000 training iterations, with model evaluation intervals of 3,000 iterations. During the fine-tuning stage, the learning rate was adjusted to 0.001, and iteration numbers and evaluation intervals were adjusted according to different novel classes. During 3-shot, 5-shot, and 10-shot training, the IoU threshold for positive samples was set to 0.6, 0.7, and 0.8, respectively, the temperature coefficient was set to 0.2, and contrastive loss weights were set to 0.2, 0.5, and 0.5 respectively.</p>
</sec>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Evaluation indicators</title>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>Evaluation criteria</title>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq17">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<inline-formula>
<mml:math display="inline" id="im57">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im58">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math display="inline" id="im59">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent true positive, false positive, and false negative, respectively. Precision and recall are defined as <xref ref-type="disp-formula" rid="eq16">Equations 16</xref>, <xref ref-type="disp-formula" rid="eq17">17</xref>, respectively.</p>
<p>When the sum of IoU between the predicted box and the target box exceeds 0.5, the predicted box is positive, otherwise it is negative.</p>
<disp-formula id="eq18">
<label>(18)</label>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mi>0</mml:mi>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<italic>AP</italic> represents the area below the precision-recall curve, calculated as shown in <xref ref-type="disp-formula" rid="eq18">Equation 18</xref>, with accuracy as the ordinate and recall as the abscissa.</p>
<p>In FSOD, base class performance is typically measured using bAP, while nAP is used to assess the performance of novel classes. Suppose class <inline-formula>
<mml:math display="inline" id="im60">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> belongs to base classes, and class <inline-formula>
<mml:math display="inline" id="im61">
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> belongs to novel classes (<italic>N</italic> denotes the number of the training classes), <inline-formula>
<mml:math display="inline" id="im62">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im63">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be expressed by <xref ref-type="disp-formula" rid="eq19">Equations 19</xref>, <xref ref-type="disp-formula" rid="eq20">20</xref>.</p>
<disp-formula id="eq19">
<label>(19)</label>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msubsup>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq20">
<label>(20)</label>
<mml:math display="block" id="M20">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>B</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the subsequent analysis, we also utilize mAP, expressed by <xref ref-type="disp-formula" rid="eq21">Equation 21</xref>.</p>
<disp-formula id="eq21">
<label>(21)</label>
<mml:math display="block" id="M21">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>c</italic> represents the class, <italic>n</italic> represents the number of classes, and <inline-formula>
<mml:math display="inline" id="im64">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the average <inline-formula>
<mml:math display="inline" id="im65">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of multiple classes. The overall effect of multi-class target detection can be represented by <inline-formula>
<mml:math display="inline" id="im66">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Comparative analysis of experiments</title>
<p>Our method will be compared with several classic FSOD methods, including the classic fine-tuning method TFA (<xref ref-type="bibr" rid="B40">Wang et&#xa0;al., 2020</xref>), the feature aggregation method based on the meta-learning framework VFA (<xref ref-type="bibr" rid="B13">Han et&#xa0;al., 2023</xref>), and the two-stage learning method (<xref ref-type="bibr" rid="B35">Sun et&#xa0;al., 2021</xref>) based on contrastive learning. Additionally, we incorporate YOLO (<xref ref-type="bibr" rid="B16">Khanam and Hussain, 2024</xref>), a widely adopted one-stage object detection framework that is particularly known for its real-time performance in various detection tasks. YOLO (You Only Look Once) significantly differs from two-stage models like Faster R-CNN by integrating region proposal and classification into a single, unified network, making it highly efficient and fast for both training and inference. All comparative experiments are trained and tested on the MMFewShot framework produced by Open MMLab. Our model&#x2019;s indicators are significantly better than most of the most advanced SOTA methods.</p>
<sec id="s4_3_1">
<label>4.3.1</label>
<title>Analysis of basic stage results</title>
<p>
<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> shows the overall loss curves of the three class split sets (split 1, 2, 3) during basic stage training. As can be seen from the figure, the loss curves of TFA, FSCE and <bold>the proposed method (OURS)</bold> are almost completely overlapped, indicating that the learning process of the three methods in the basic training stage is very similar. Since the variational autoencoder introduces additional loss terms during training, the loss of VFA is higher. Overall, the loss of the four methods is gradually decreasing with the increase in the number of training iterations, indicating that the model is constantly learning and improving.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Line chart of overall loss of basic stage training.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g003.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> shows the changes in mAP50 (average accuracy under IoU 0.5) of four different class splits during basic stage, and the test set is evaluated every 3000 iterations. In split1, VFA, FSCE, and the proposed method reach maximum mAP50 values of 86.6, 88.9, and 89.7 at 18,000 iterations, respectively; while TFA reaches a maximum mAP50 value of 89.1 at 15,000 iterations, but drops at 18,000 iterations, indicating possible overfitting. In split2, TFA and VFA reach maximum mAP50 values &#x200b;&#x200b;of 88.6 and 85.6 at 18,000 iterations, respectively, while FSCE and the proposed method reach 89.0 and 89.8 at 15,000 iterations, and also show overfitting at 18,000 iterations. In split3, TFA and FSCE reached 85.8 and 85.3 respectively at 18,000 iterations, while VFA and the proposed method reached the maximum value of 84.5 and 85.9 at 15,000 iterations, but overfitting also occurred at 18,000 iterations. These results reflect the differences in the sensitivity of different methods to the number of training iterations and the stability in the later stages of training.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Basic stage testing mAP50 indicator line chart.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g004.tif"/>
</fig>
<p>Based on these results, we will adopt the following strategy for subsequent fine-tuning: selecting the models saved at the point where mAP50 achieves the highest value in splits 1 to 3 as the starting point for fine-tuning. the proposed method is to leverage the model state that achieves optimal performance during the base stage to further enhance its performance in the few-shot object detection task.</p>
</sec>
<sec id="s4_3_2">
<label>4.3.2</label>
<title>Fine-tuning experimental results analysis</title>
<p>
<xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5</bold>
</xref>&#x2013;<xref ref-type="fig" rid="f7">
<bold>7</bold>
</xref> show the visualization results of the relevant data after two rounds of random sampling and fine-tuning, and the results on the test set with different sample numbers (3, 5, and 10), covering splits 1, 2, and 3. The performance of each method (TFA, VFA, FSCE, and the proposed method) is measured by the average precision of the base class (bAP50), the average precision of the new class (nAP50), and the overall average precision (mAP50).As can be seen from <xref ref-type="fig" rid="f5">
<bold>Figures&#xa0;5</bold>
</xref>&#x2013;<xref ref-type="fig" rid="f7">
<bold>7</bold>
</xref>, TFA and VFA show an inverse relationship in performance: TFA performs well on the base class (bAP50), but is relatively weak on the new class (nAP50), which indicates that TFA may not be able to effectively transfer knowledge to the new class. In contrast, VFA performs well in the new class but poorly in the base class, which indicates that the model may sacrifice the performance of the base class to adapt to the new class. In contrast, FSCE performs evenly in the two classes and shows better robustness. the proposed method performs better on the basis of FSCE. Under certain split and shot configurations, the proposed method even slightly outperforms VFA in terms of new classes and overall accuracy, indicating its excellent ability in balancing the performance difference between base and new classes.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Trends of bAP50, nAP50, and mAP50 for split1.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g005.tif"/>
</fig>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Trends of bAP50, nAP50, and mAP50 for split2.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g006.tif"/>
</fig>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Trends of bAP50, nAP50 and mAP50 for split3.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g007.tif"/>
</fig>
<p>In the nAP50 graph of new classes for split3, VFA outperforms other methods under 3-shot conditions; but its performance improves only slightly with the increase in sample size, increasing by only 11.04% from 3-shot to 10-shot. In contrast, the performance of the proposed method improves significantly, increasing by 23.95% from 3-shot to 10-shot. To further study this phenomenon, a third random sampling fine-tuning training experiment was conducted based on the split3 dataset.</p>
<p>
<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> shows the changing trends of bAP50 and nAP50 during split3 fine-tuning training under different shot conditions. The performance of FSCE and the proposed method in bAP50 is always between TFA and VFA, but its maximum nAP50 exceeds that of other methods, which highlights the advantage of the proposed method in balancing the performance of base and new classes.</p>
<p>To comprehensively compare the performance of detection methods across 20 tropical pest classes, we included the YOLO model (specifically the YOLO11x version) as a benchmark for FSOD tasks. <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> and relevant results incorporate the YOLO method alongside TFA, VFA, FSCE, and the proposed method (OURS). While YOLO is known for its efficiency in real-time detection tasks due to lower computational complexity, the results indicate that this advantage does not translate into better performance in FSOD scenarios. The results show that YOLO, TFA, VFA, FSCE and the proposed method all show high AP values &#x200b;&#x200b;under 3-shot, 5-shot and 10-shot conditions, proving the stability of the methods. With the increase of sample size, the performance of the three methods in new classes gradually improves, especially when the sample size is small, the detection performance is significantly improved with a slight increase in sample size.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>AP values and mAP values of four methods for detecting 20 types of pests.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Cat.</th>
<th valign="middle" rowspan="2" align="center">Num.</th>
<th valign="top" colspan="3" align="center">YOLO11-FSOD</th>
<th valign="middle" colspan="3" align="center">TFA</th>
<th valign="middle" colspan="3" align="center">VFA</th>
<th valign="middle" colspan="3" align="center">FSCE</th>
<th valign="middle" colspan="3" align="center">OURS</th>
</tr>
<tr>
<th valign="middle" align="center">3shot</th>
<th valign="middle" align="center">5shot</th>
<th valign="middle" align="center">10shot</th>
<th valign="middle" align="center">3shot</th>
<th valign="middle" align="center">5shot</th>
<th valign="middle" align="center">10shot</th>
<th valign="middle" align="center">3shot</th>
<th valign="middle" align="center">5shot</th>
<th valign="middle" align="center">10shot</th>
<th valign="middle" align="center">3shot</th>
<th valign="middle" align="center">5shot</th>
<th valign="middle" align="center">10shot</th>
<th valign="middle" align="center">3shot</th>
<th valign="middle" align="center">5shot</th>
<th valign="middle" align="center">10shot</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="15" align="center">Base class<break/>AP</td>
<td valign="middle" align="center">0</td>
<td valign="middle" align="center">82.5</td>
<td valign="middle" align="center">84.4</td>
<td valign="middle" align="center">87</td>
<td valign="middle" align="center">96</td>
<td valign="middle" align="center">96</td>
<td valign="middle" align="center">95.8</td>
<td valign="middle" align="center">90.2</td>
<td valign="middle" align="center">97.9</td>
<td valign="middle" align="center">90.4</td>
<td valign="middle" align="center">91.7</td>
<td valign="middle" align="center">92.1</td>
<td valign="middle" align="center">92</td>
<td valign="middle" align="center">90.7</td>
<td valign="middle" align="center">89.8</td>
<td valign="middle" align="center">96.2</td>
</tr>
<tr>
<td valign="middle" align="center">48</td>
<td valign="middle" align="center">59.6</td>
<td valign="middle" align="center">61.2</td>
<td valign="middle" align="center">57.9</td>
<td valign="middle" align="center">68.5</td>
<td valign="middle" align="center">69</td>
<td valign="middle" align="center">69.8</td>
<td valign="middle" align="center">63.4</td>
<td valign="middle" align="center">67</td>
<td valign="middle" align="center">69.2</td>
<td valign="middle" align="center">65.6</td>
<td valign="middle" align="center">68</td>
<td valign="middle" align="center">66.6</td>
<td valign="middle" align="center">67</td>
<td valign="middle" align="center">66.7</td>
<td valign="middle" align="center">63.6</td>
</tr>
<tr>
<td valign="middle" align="center">14</td>
<td valign="middle" align="center">83</td>
<td valign="middle" align="center">89.7</td>
<td valign="middle" align="center">86.2</td>
<td valign="middle" align="center">95.4</td>
<td valign="middle" align="center">92.9</td>
<td valign="middle" align="center">89.5</td>
<td valign="middle" align="center">88.4</td>
<td valign="middle" align="center">88.7</td>
<td valign="middle" align="center">88.5</td>
<td valign="middle" align="center">95.9</td>
<td valign="middle" align="center">94.5</td>
<td valign="middle" align="center">88</td>
<td valign="middle" align="center">89.2</td>
<td valign="middle" align="center">87.7</td>
<td valign="middle" align="center">88.3</td>
</tr>
<tr>
<td valign="middle" align="center">99</td>
<td valign="middle" align="center">80.8</td>
<td valign="middle" align="center">75.3</td>
<td valign="middle" align="center">68.9</td>
<td valign="middle" align="center">85.2</td>
<td valign="middle" align="center">85</td>
<td valign="middle" align="center">84.8</td>
<td valign="middle" align="center">86.8</td>
<td valign="middle" align="center">88.2</td>
<td valign="middle" align="center">89.1</td>
<td valign="middle" align="center">75.4</td>
<td valign="middle" align="center">77.5</td>
<td valign="middle" align="center">82.5</td>
<td valign="middle" align="center">78.4</td>
<td valign="middle" align="center">81.7</td>
<td valign="middle" align="center">76.7</td>
</tr>
<tr>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center">85.9</td>
<td valign="middle" align="center">77</td>
<td valign="middle" align="center">79.3</td>
<td valign="middle" align="center">69.6</td>
<td valign="middle" align="center">71.3</td>
<td valign="middle" align="center">71.4</td>
<td valign="middle" align="center">72.6</td>
<td valign="middle" align="center">77.4</td>
<td valign="middle" align="center">75</td>
<td valign="middle" align="center">66.3</td>
<td valign="middle" align="center">74.3</td>
<td valign="middle" align="center">71.5</td>
<td valign="middle" align="center">69.9</td>
<td valign="middle" align="center">76.1</td>
<td valign="middle" align="center">80.3</td>
</tr>
<tr>
<td valign="middle" align="center">21</td>
<td valign="middle" align="center">77.6</td>
<td valign="middle" align="center">80.5</td>
<td valign="middle" align="center">83.8</td>
<td valign="middle" align="center">89</td>
<td valign="middle" align="center">88.8</td>
<td valign="middle" align="center">88.8</td>
<td valign="middle" align="center">74.7</td>
<td valign="middle" align="center">72.8</td>
<td valign="middle" align="center">73.7</td>
<td valign="middle" align="center">85.3</td>
<td valign="middle" align="center">80.9</td>
<td valign="middle" align="center">81</td>
<td valign="middle" align="center">87.8</td>
<td valign="middle" align="center">87.3</td>
<td valign="middle" align="center">86.7</td>
</tr>
<tr>
<td valign="middle" align="center">39</td>
<td valign="middle" align="center">84.8</td>
<td valign="middle" align="center">76.4</td>
<td valign="middle" align="center">79.6</td>
<td valign="middle" align="center">88</td>
<td valign="middle" align="center">88</td>
<td valign="middle" align="center">93.3</td>
<td valign="middle" align="center">88.2</td>
<td valign="middle" align="center">88.4</td>
<td valign="middle" align="center">89.3</td>
<td valign="middle" align="center">83.8</td>
<td valign="middle" align="center">85.7</td>
<td valign="middle" align="center">86.6</td>
<td valign="middle" align="center">89.8</td>
<td valign="middle" align="center">89.1</td>
<td valign="middle" align="center">89.3</td>
</tr>
<tr>
<td valign="middle" align="center">66</td>
<td valign="middle" align="center">95.1</td>
<td valign="middle" align="center">96.4</td>
<td valign="middle" align="center">97.8</td>
<td valign="middle" align="center">89.7</td>
<td valign="middle" align="center">90</td>
<td valign="middle" align="center">89.9</td>
<td valign="middle" align="center">89.1</td>
<td valign="middle" align="center">89.1</td>
<td valign="middle" align="center">89.1</td>
<td valign="middle" align="center">92.9</td>
<td valign="middle" align="center">93.5</td>
<td valign="middle" align="center">89.9</td>
<td valign="middle" align="center">94.9</td>
<td valign="middle" align="center">90.4</td>
<td valign="middle" align="center">90.7</td>
</tr>
<tr>
<td valign="middle" align="center">16</td>
<td valign="middle" align="center">67.2</td>
<td valign="middle" align="center">72.8</td>
<td valign="middle" align="center">68.9</td>
<td valign="middle" align="center">84.3</td>
<td valign="middle" align="center">84.4</td>
<td valign="middle" align="center">80.9</td>
<td valign="middle" align="center">67.7</td>
<td valign="middle" align="center">66.5</td>
<td valign="middle" align="center">70.7</td>
<td valign="middle" align="center">78.8</td>
<td valign="middle" align="center">80.2</td>
<td valign="middle" align="center">79.4</td>
<td valign="middle" align="center">81.4</td>
<td valign="middle" align="center">79.3</td>
<td valign="middle" align="center">83.8</td>
</tr>
<tr>
<td valign="middle" align="center">37</td>
<td valign="middle" align="center">79.2</td>
<td valign="middle" align="center">77.4</td>
<td valign="middle" align="center">71.8</td>
<td valign="middle" align="center">99.4</td>
<td valign="middle" align="center">99.3</td>
<td valign="middle" align="center">97.8</td>
<td valign="middle" align="center">88</td>
<td valign="middle" align="center">88.1</td>
<td valign="middle" align="center">88.4</td>
<td valign="middle" align="center">89</td>
<td valign="middle" align="center">89.6</td>
<td valign="middle" align="center">93.8</td>
<td valign="middle" align="center">88.9</td>
<td valign="middle" align="center">89.1</td>
<td valign="middle" align="center">90.4</td>
</tr>
<tr>
<td valign="middle" align="center">50</td>
<td valign="middle" align="center">82.8</td>
<td valign="middle" align="center">73.8</td>
<td valign="middle" align="center">80.7</td>
<td valign="middle" align="center">85.1</td>
<td valign="middle" align="center">85.5</td>
<td valign="middle" align="center">86.1</td>
<td valign="middle" align="center">86.5</td>
<td valign="middle" align="center">87.4</td>
<td valign="middle" align="center">88.1</td>
<td valign="middle" align="center">83</td>
<td valign="middle" align="center">84.1</td>
<td valign="middle" align="center">84.5</td>
<td valign="middle" align="center">83.8</td>
<td valign="middle" align="center">79.1</td>
<td valign="middle" align="center">85.9</td>
</tr>
<tr>
<td valign="middle" align="center">26</td>
<td valign="middle" align="center">80.7</td>
<td valign="middle" align="center">80</td>
<td valign="middle" align="center">81.6</td>
<td valign="middle" align="center">84.6</td>
<td valign="middle" align="center">83.7</td>
<td valign="middle" align="center">83.1</td>
<td valign="middle" align="center">66.6</td>
<td valign="middle" align="center">71.6</td>
<td valign="middle" align="center">78.6</td>
<td valign="middle" align="center">80.3</td>
<td valign="middle" align="center">77.7</td>
<td valign="middle" align="center">78.8</td>
<td valign="middle" align="center">84.8</td>
<td valign="middle" align="center">80.4</td>
<td valign="middle" align="center">85.7</td>
</tr>
<tr>
<td valign="middle" align="center">25</td>
<td valign="middle" align="center">82.8</td>
<td valign="middle" align="center">83.3</td>
<td valign="middle" align="center">86.4</td>
<td valign="middle" align="center">89</td>
<td valign="middle" align="center">89.3</td>
<td valign="middle" align="center">89.2</td>
<td valign="middle" align="center">81.8</td>
<td valign="middle" align="center">84.6</td>
<td valign="middle" align="center">88.8</td>
<td valign="middle" align="center">89.2</td>
<td valign="middle" align="center">89.3</td>
<td valign="middle" align="center">88.4</td>
<td valign="middle" align="center">87.9</td>
<td valign="middle" align="center">88.6</td>
<td valign="middle" align="center">90.5</td>
</tr>
<tr>
<td valign="middle" align="center">70</td>
<td valign="middle" align="center">63.2</td>
<td valign="middle" align="center">54.4</td>
<td valign="middle" align="center">57.9</td>
<td valign="middle" align="center">67.8</td>
<td valign="middle" align="center">67.7</td>
<td valign="middle" align="center">66.1</td>
<td valign="middle" align="center">52.8</td>
<td valign="middle" align="center">59.3</td>
<td valign="middle" align="center">66.2</td>
<td valign="middle" align="center">68.8</td>
<td valign="middle" align="center">73.2</td>
<td valign="middle" align="center">71.5</td>
<td valign="middle" align="center">73.6</td>
<td valign="middle" align="center">70.6</td>
<td valign="middle" align="center">62.8</td>
</tr>
<tr>
<td valign="middle" align="center">24</td>
<td valign="middle" align="center">85.8</td>
<td valign="middle" align="center">75.4</td>
<td valign="middle" align="center">79.5</td>
<td valign="middle" align="center">87.6</td>
<td valign="middle" align="center">86.7</td>
<td valign="middle" align="center">87.7</td>
<td valign="middle" align="center">84.9</td>
<td valign="middle" align="center">84.5</td>
<td valign="middle" align="center">84.5</td>
<td valign="middle" align="center">78.4</td>
<td valign="middle" align="center">76.5</td>
<td valign="middle" align="center">83.8</td>
<td valign="middle" align="center">84.8</td>
<td valign="middle" align="center">84.1</td>
<td valign="middle" align="center">85.7</td>
</tr>
<tr>
<td valign="middle" rowspan="5" align="center">New class<break/>AP</td>
<td valign="middle" align="center">67</td>
<td valign="middle" align="center">72.7</td>
<td valign="middle" align="center">98.2</td>
<td valign="middle" align="center">98.5</td>
<td valign="middle" align="center">81.8</td>
<td valign="middle" align="center">82.7</td>
<td valign="middle" align="center">83.3</td>
<td valign="middle" align="center">92.1</td>
<td valign="middle" align="center">94</td>
<td valign="middle" align="center">96.5</td>
<td valign="middle" align="center">90.9</td>
<td valign="middle" align="center">90.1</td>
<td valign="middle" align="center">94.4</td>
<td valign="middle" align="center">90.9</td>
<td valign="middle" align="center">95</td>
<td valign="middle" align="center">97.6</td>
</tr>
<tr>
<td valign="middle" align="center">101</td>
<td valign="middle" align="center">63.9</td>
<td valign="middle" align="center">81.6</td>
<td valign="middle" align="center">88.8</td>
<td valign="middle" align="center">17</td>
<td valign="middle" align="center">25.7</td>
<td valign="middle" align="center">49</td>
<td valign="middle" align="center">38.5</td>
<td valign="middle" align="center">55.3</td>
<td valign="middle" align="center">67.7</td>
<td valign="middle" align="center">18.7</td>
<td valign="middle" align="center">36.9</td>
<td valign="middle" align="center">66.7</td>
<td valign="middle" align="center">35.9</td>
<td valign="middle" align="center">65</td>
<td valign="middle" align="center">82.5</td>
</tr>
<tr>
<td valign="middle" align="center">76</td>
<td valign="middle" align="center">17.3</td>
<td valign="middle" align="center">23.8</td>
<td valign="middle" align="center">36.8</td>
<td valign="middle" align="center">12</td>
<td valign="middle" align="center">27.8</td>
<td valign="middle" align="center">28.8</td>
<td valign="middle" align="center">17.2</td>
<td valign="middle" align="center">28.8</td>
<td valign="middle" align="center">33.7</td>
<td valign="middle" align="center">28.3</td>
<td valign="middle" align="center">33.1</td>
<td valign="middle" align="center">34.6</td>
<td valign="middle" align="center">30.1</td>
<td valign="middle" align="center">36.8</td>
<td valign="middle" align="center">44.9</td>
</tr>
<tr>
<td valign="middle" align="center">15</td>
<td valign="middle" align="center">51.4</td>
<td valign="middle" align="center">85.9</td>
<td valign="middle" align="center">89.4</td>
<td valign="middle" align="center">65</td>
<td valign="middle" align="center">64.8</td>
<td valign="middle" align="center">75.2</td>
<td valign="middle" align="center">66.6</td>
<td valign="middle" align="center">83.2</td>
<td valign="middle" align="center">88.3</td>
<td valign="middle" align="center">68.9</td>
<td valign="middle" align="center">79.8</td>
<td valign="middle" align="center">89.7</td>
<td valign="middle" align="center">69</td>
<td valign="middle" align="center">77.9</td>
<td valign="middle" align="center">87.7</td>
</tr>
<tr>
<td valign="middle" align="center">95</td>
<td valign="middle" align="center">60.5</td>
<td valign="middle" align="center">79.3</td>
<td valign="middle" align="center">72.4</td>
<td valign="middle" align="center">51.2</td>
<td valign="middle" align="center">58.6</td>
<td valign="middle" align="center">64.4</td>
<td valign="middle" align="center">42.5</td>
<td valign="middle" align="center">38.4</td>
<td valign="middle" align="center">57.4</td>
<td valign="middle" align="center">67.5</td>
<td valign="middle" align="center">68.1</td>
<td valign="middle" align="center">72.8</td>
<td valign="middle" align="center">74.1</td>
<td valign="middle" align="center">76.1</td>
<td valign="middle" align="center">83.5</td>
</tr>
<tr>
<td valign="middle" rowspan="3" align="center">mAP</td>
<td valign="middle" align="center">Base class</td>
<td valign="middle" align="center">79.4</td>
<td valign="middle" align="center">77.2</td>
<td valign="middle" align="center">77.8</td>
<td valign="middle" align="center">85.2</td>
<td valign="middle" align="center">85.1</td>
<td valign="middle" align="center">84.9</td>
<td valign="middle" align="center">78.7</td>
<td valign="middle" align="center">80.7</td>
<td valign="middle" align="center">84.9</td>
<td valign="middle" align="center">81.6</td>
<td valign="middle" align="center">82.4</td>
<td valign="middle" align="center">82.5</td>
<td valign="middle" align="center">83.5</td>
<td valign="middle" align="center">82.6</td>
<td valign="middle" align="center">83.7</td>
</tr>
<tr>
<td valign="middle" align="center">New class</td>
<td valign="middle" align="center">53.2</td>
<td valign="middle" align="center">73.8</td>
<td valign="middle" align="center">77.2</td>
<td valign="middle" align="center">45.3</td>
<td valign="middle" align="center">51.9</td>
<td valign="middle" align="center">60.1</td>
<td valign="middle" align="center">51.3</td>
<td valign="middle" align="center">59.9</td>
<td valign="middle" align="center">60.1</td>
<td valign="middle" align="center">54.8</td>
<td valign="middle" align="center">61.8</td>
<td valign="middle" align="center">71.6</td>
<td valign="middle" align="center">60</td>
<td valign="middle" align="center">70.1</td>
<td valign="middle" align="center">79.2</td>
</tr>
<tr>
<td valign="middle" align="center">All class</td>
<td valign="middle" align="center">66.3</td>
<td valign="middle" align="center">75.5</td>
<td valign="middle" align="center">77.5</td>
<td valign="middle" align="center">75.3</td>
<td valign="middle" align="center">76.9</td>
<td valign="middle" align="center">78.7</td>
<td valign="middle" align="center">71.9</td>
<td valign="middle" align="center">75.6</td>
<td valign="middle" align="center">78.7</td>
<td valign="middle" align="center">74.9</td>
<td valign="middle" align="center">77.2</td>
<td valign="middle" align="center">79.8</td>
<td valign="middle" align="center">75.9</td>
<td valign="middle" align="center">79.5</td>
<td valign="middle" align="center">82.6</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In terms of detection performance in each class, the proposed method shows an upward or stable trend in mAP value with the increase of sample size, while YOLO, TFA, VFA and FSCE have certain fluctuations. Especially in the new class, the proposed method achieved the maximum mAP value of 82% in the 10-shot experiment, which is significantly better than YOLO, TFA, VFA and FSCE. In addition, the proposed method shows particularly excellent performance in specific classes such as 15 and 95, and significantly improves AP in the challenging 101 class (Cicadellidae). Compared with other methods, its mAP value is nearly 3 times higher, reflecting the powerful feature aggregation and migration capabilities of the proposed method.</p>
<p>Although the mAP values &#x200b;&#x200b;of most classes are above 70, indicating that the proposed method can effectively detect these pests, the mAP values &#x200b;&#x200b;of other methods are relatively low for classes such as 48, 101, 76, and 95. As can be seen from the relevant images in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, the visual features of these pests are highly similar to the background, or have features that are difficult to distinguish from other classes, making it difficult for YOLO, TFA, VFA, and FSCE methods to accurately identify them. Overall, the mAP value of the proposed method in the new class is nearly 10 percentage points higher than that of YOLO, TFA, VFA, and FSCE on average, showing its significant advantage in the tropical pest detection task.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Visual comparison of original image RAW, YOLO, TFA, VFA, FSCE and OURS.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g008.tif"/>
</fig>
<p>In terms of detection performance under 3-shot, 5-shot, and 10-shot conditions, YOLO demonstrates relatively lower AP values compared to the other methods. For example, in the 10-shot experiment, YOLO achieves an mAP of 77.5%, whereas the proposed method achieves a significantly higher mAP of 82.6%. Notably, in challenging classes such as 48 and 101, YOLO struggles to distinguish pests with features similar to the background, resulting in mAP values below 60%, significantly lower than the corresponding performance of the proposed method. Overall, while YOLO provides a computationally efficient solution, the trade-off between speed and accuracy limits its applicability in FSOD tasks that prioritize precise detection over real-time processing. The proposed method strikes a better balance by achieving state-of-the-art detection performance, justifying the slightly higher computational cost for critical applications like pest management in tropical agricultural settings.</p>
<p>As shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, from the 10-shot confusion matrix analysis of TFA, VFA, FSCE and the proposed method in split3, the proposed method has obvious advantages in terms of accuracy, missed detection rate and recall rate. First, in terms of accuracy, the proposed method presents higher values &#x200b;&#x200b;on the diagonal, indicating that the model has higher classification accuracy on multiple classes. In contrast, TFA and FSCE methods have lower diagonal accuracy in some classes, showing that the recognition of some classes is not accurate enough under few-sample conditions. In particular, the TFA method has serious misclassification in some classes, while the proposed method is relatively balanced in overall accuracy. In addition, VFA has some misclassification in the background class, while the proposed method is better at distinguishing between targets and backgrounds. Secondly, in terms of missed detection rate performance, the off-diagonal misclassification rate of the proposed method is lower, which means that it has fewer missed detections. In contrast, the FSCE and VFA methods have high missed detection rates in some classes, especially between difficult-to-distinguish classes, which are prone to prediction deviation. FSCE has more obvious misclassification in medium-complexity classes, while VFA shows a tendency to misdetect when the background interference is strong, resulting in an increase in missed detection rate. the proposed method significantly reduces the missed detection rate and improves overall reliability by improving feature extraction. Finally, in terms of recall rate, the proposed method has a higher recall rate in most classes. With fewer misclassifications, the proposed method can effectively identify more real samples, especially in complex backgrounds or with few samples, and the recall performance is more stable. In contrast, the recall rate of the TFA method is low, and it is easy to make recognition errors when the class boundaries are blurred. The recall rate of FSCE is also slightly insufficient when dealing with some subdivided classes.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Confusion matrix of split3-10shot for TFA <bold>(a)</bold>, VFA <bold>(b)</bold>, FSCE <bold>(c)</bold> and OURS <bold>(d)</bold>.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g009.tif"/>
</fig>
<p>In summary, the proposed method is superior to other methods in accuracy, missed detection rate and recall rate. Its advantages lie in better feature extraction ability, lower misclassification rate and higher recall rate, making it a more robust model in the case of few samples and complex backgrounds. These improvements enable the proposed method to perform better classification results in the split3 10-shot scenario.</p>
</sec>
<sec id="s4_3_3">
<label>4.3.3</label>
<title>Ablation experiment analysis</title>
<p>We have evaluated the effectiveness of the modules used in the study, such as the feature aggregation module(FAM), the SCL module SCL, and the multi-task loss optimization MTLF, in detail through ablation experiments. In the ablation study show in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, we systematically introduced three key modules based on the baseline method TFA. By incorporating these modules into the baseline separately, we conducted 3-shot, 5-shot, and 10-shot experiments in the few-shot scenario, and evaluated them in terms of bAP50, nAP50, and mAP50. Through this comprehensive evaluation, we can thoroughly investigate and verify the effectiveness of each component in the framework, which helps to further fully understand the proposed method. The ablation results are shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, showing the effect of the key modules. The performance is significantly improved by about 1.1% by introducing FAM alone. Specifically, mAP increases from 0.741 to 0.751 in the case of 3 shots, from 0.772 to 0.779 in the case of 5 shots, and from 0.794 to 0.805 in the case of 10 shots. In addition, the inclusion of the SCL module alone can improve its performance by about 1.5%. In the case of 3 shots, mAP increases from 0.741 to 0.753, in the case of 5 shots, from 0.772 to 0.787, and in the case of 10 shots, from 0.794 to 0.809, highlighting the effectiveness of the SCL module in addressing the multi-scale challenges encountered in pest object detection. In addition, adopting the multi-task loss optimization module as a standalone ensemble on the baseline improves the results by about 3.5%. This improvement is evident in the case of 3 shots, where mAP increases from 0.741 to 0.776, in the case of 5 shots, from 0.772 to 0.807, and in the case of 10 shots, from 0.794 to 0.826.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Ablation experiment.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g010.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> shows an example of the comparison of the new class detection results of the proposed method with those of TFA, VFA, and FSCE methods in the dataset. As shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, most of the new class objects are correctly detected, demonstrating the efficiency of our model. Other methods have difficulty in effectively detecting new class multi-target situations. In <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, we can see that although the insect is similar to the background, our model correctly identifies the background and does not misidentify the insect. Similarly, although there are multiple insect targets in the image, our model can still correctly identify all the targets. Our model can effectively handle size variations and multiple targets, and correctly identify single targets and multiple targets of different sizes. Edge cases, such as overlapping pests or those camouflaged within cluttered backgrounds, posed challenges for all tested models. While the proposed method outperformed others in these scenarios, future work could explore adaptive feature learning techniques or advanced data preprocessing to further improve performance in such cases&#x201d;.</p>
</sec>
<sec id="s4_3_4">
<label>4.3.4</label>
<title>Model statistical characteristics analysis</title>
<p>To evaluate the stability and differences of the proposed method compared to other methods, statistical analysis and significance tests were conducted. As shown in <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>, our method outperforms the comparison methods (TFA, VFA, and FSCE) in terms of statistical metrics such as mean (Mean), standard deviation (Std), and confidence interval (CI). The mean value of OURS is 79.13, which is higher than TFA (75.30), VFA (75.07), and FSCE (77.26), indicating its superior overall performance. Furthermore, the standard deviation of the proposed method is 1.920, lower than those of VFA (2.471) and FSCE (2.233), demonstrating greater stability. Within the 95% confidence interval, the proposed method exhibits a range of (78.178, 80.088), which is significantly higher than the intervals of other methods, such as TFA (74.541, 76.059). This indicates that the proposed method holds a clear statistical advantage.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Mean, standard deviation and confidence interval statistical analysis.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Mean</th>
<th valign="middle" align="center">Standard deviation (Std)</th>
<th valign="middle" align="center">Standard error (SE)</th>
<th valign="middle" align="center">Margin of error (MOE)</th>
<th valign="middle" align="center">95% Confidence interval (CI)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">TFA</td>
<td valign="middle" align="left">75.30</td>
<td valign="middle" align="left">1.526</td>
<td valign="middle" align="left">0.36</td>
<td valign="middle" align="left">0.759</td>
<td valign="middle" align="left">(74.541, 76.059)</td>
</tr>
<tr>
<td valign="middle" align="left">FSCE</td>
<td valign="middle" align="left">77.26</td>
<td valign="middle" align="left">2.233</td>
<td valign="middle" align="left">0.526</td>
<td valign="middle" align="left">1.111</td>
<td valign="middle" align="left">(76.151, 78.372)</td>
</tr>
<tr>
<td valign="middle" align="left">VFA</td>
<td valign="middle" align="left">75.07</td>
<td valign="middle" align="left">2.471</td>
<td valign="middle" align="left">0.583</td>
<td valign="middle" align="left">1.229</td>
<td valign="middle" align="left">(73.843, 76.301)</td>
</tr>
<tr>
<td valign="middle" align="left">OURS</td>
<td valign="middle" align="left">79.13</td>
<td valign="middle" align="left">1.92</td>
<td valign="middle" align="left">0.453</td>
<td valign="middle" align="left">0.955</td>
<td valign="middle" align="left">(78.178, 80.088)</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As presented in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>, statistical significance tests based on multiple independent experimental results further confirm the advantages of the proposed method compared to TFA, VFA, and FSCE. Using independent t-tests at a significance level of 0.05, the results show that the p-value for the proposed method versus TFA is 0.00116, versus FSCE is 0.03284, and versus VFA is 0.00288&#x2014;all below 0.05. This demonstrates that the performance of the proposed method is statistically significantly different from the other models. Additionally, the mean value of OURS is 79.13, which is higher than TFA (77.05), FSCE (77.62), and VFA (76.87). These results indicate that the proposed method not only outperforms other models in overall performance but also achieves statistically significant differences across multiple experiments. In summary, the proposed method demonstrates superior stability and performance compared to other models, highlighting its statistical advantages.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Independent t-test method significance verification analysis.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Comparison model</th>
<th valign="middle" align="center">Significance level</th>
<th valign="middle" align="center">P-value</th>
<th valign="middle" align="center">Model mean</th>
<th valign="middle" align="center">Comparison model mean</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">OURS</td>
<td valign="middle" align="center">TFA</td>
<td valign="middle" align="center">0.05</td>
<td valign="middle" align="center">0.00116</td>
<td valign="middle" align="center">79.13</td>
<td valign="middle" align="center">77.05</td>
</tr>
<tr>
<td valign="middle" align="center">OURS</td>
<td valign="middle" align="center">FSCE</td>
<td valign="middle" align="center">0.05</td>
<td valign="middle" align="center">0.03284</td>
<td valign="middle" align="center">79.13</td>
<td valign="middle" align="center">77.62</td>
</tr>
<tr>
<td valign="middle" align="center">OURS</td>
<td valign="middle" align="center">VFA</td>
<td valign="middle" align="center">0.05</td>
<td valign="middle" align="center">0.00288</td>
<td valign="middle" align="center">79.13</td>
<td valign="middle" align="center">76.87</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_3_5">
<label>4.3.5</label>
<title>Computational cost and performance trade-off analysis</title>
<p>The analysis of computational complexity and detection performance highlights the trade-offs made in this study. YOLO, known for its efficiency in real-time detection tasks, achieves the lowest computational complexity with 114.5 GFLOPs and relatively moderate mAP values (66.3, 75.5, 77.5 for 3-shot, 5-shot, and 10-shot tasks, respectively). In contrast, OURS, a model based on the Faster R-CNN framework with enhancements such as FAM and SCL, achieves the highest mAP values across all settings (75.9, 79.5, 82.6) at a slightly higher computational cost of 130.2 GFLOPs. These results, summarized in <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref>, clearly demonstrate the performance and computational trade-offs between YOLO and OURS. This demonstrates that OURS leverages the computational resources to achieve significant performance gains, particularly in few-shot detection tasks, where accuracy and robustness are critical. While YOLO is more suitable for real-time applications, its lower performance in few-shot tasks highlights its limitations in capturing fine-grained and diverse pest characteristics. Models like TFA, FSCE, and VFA strike a balance between complexity and performance, but they fall short of the proposed method in overall accuracy.</p>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Model size, computational cost, and performance analysis.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Number of parameters</th>
<th valign="middle" align="center">Calculate costs (GPLOPs)</th>
<th valign="middle" align="center">3shot all class</th>
<th valign="middle" align="center">5shot all class</th>
<th valign="middle" align="center">10shot all class</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">YOLO11x-FSOD</td>
<td valign="middle" align="center">53.9M</td>
<td valign="middle" align="center">114.5</td>
<td valign="middle" align="center">66.3</td>
<td valign="middle" align="center">75.5</td>
<td valign="middle" align="center">77.5</td>
</tr>
<tr>
<td valign="middle" align="center">TFA</td>
<td valign="middle" align="center">60.4M</td>
<td valign="middle" align="center">119.6</td>
<td valign="middle" align="center">75.3</td>
<td valign="middle" align="center">76.9</td>
<td valign="middle" align="center">78.7</td>
</tr>
<tr>
<td valign="middle" align="center">FSCE</td>
<td valign="middle" align="center">61.6M</td>
<td valign="middle" align="center">120.8</td>
<td valign="middle" align="center">71.9</td>
<td valign="middle" align="center">75.6</td>
<td valign="middle" align="center">78.7</td>
</tr>
<tr>
<td valign="middle" align="center">VFA</td>
<td valign="middle" align="center">68.5M</td>
<td valign="middle" align="center">128.8</td>
<td valign="middle" align="center">74.9</td>
<td valign="middle" align="center">77.2</td>
<td valign="middle" align="center">79.8</td>
</tr>
<tr>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">68.1M</td>
<td valign="middle" align="center">130.2</td>
<td valign="middle" align="center">75.9</td>
<td valign="middle" align="center">79.5</td>
<td valign="middle" align="center">82.6</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>By choosing Faster R-CNN as the base framework, this study prioritizes higher detection accuracy over real-time speed, a trade-off that is justified for applications requiring precise pest management. This approach demonstrates that slight increases in computational complexity are acceptable to achieve substantial performance improvements, aligning with the study&#x2019;s goal of advancing few-shot object detection in complex agricultural environments.</p>
</sec>
<sec id="s4_3_6">
<label>4.3.6</label>
<title>Practical application and field validation</title>
<p>The proposed algorithm has been integrated into a practical pest management system, whose architectural design (as depicted at the top of <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>) addresses three specific application scenarios: Under weak network conditions, front-end devices with edge computing capabilities perform local pest detection in real-time and autonomously activate laser-based capture mechanisms. Under stable network conditions, low-cost front-end visual sensors transmit images to a backend cloud platform for rapid pest identification, subsequently triggering front-end laser capture devices, thus optimizing deployment costs. Agricultural technicians or unmanned aerial vehicles (UAVs) upload images to the backend platform, enabling precise identification and geolocation-based positioning, supporting flexible mobile monitoring. The backend cloud platform employs parallel computing to achieve millisecond-level processing and feedback, effectively fulfilling diverse scenario requirements and establishing a comprehensive intelligent pest management system encompassing real-time monitoring, rapid identification, precise localization, and targeted pest control.</p>
<p>As shown in the lower-left section of <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>, the pest induction and laser capture device comprises key modules including a core computing board, laser emitter, galvanometer controller, and visual sensing components. Specific attractants or optical methods accurately lure pests onto designated induction panel areas. Real-time visual data captured by onboard cameras is swiftly processed by a lightweight detection algorithm developed in this research, which can also be deployed in parallel on cloud platforms to handle large volumes of data from multiple devices simultaneously. The coordinate conversion module precisely calculates the physical positions of detected pests, guiding the laser galvanometer to accurately target and activate the laser for pest capture. Captured pests are subsequently collected in designated containers for further identification and analysis. This approach effectively minimizes environmental interference and protects beneficial insects, significantly enhancing the precision and effectiveness of pest monitoring and control.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Pest management system architecture, laser trapping equipment and process demonstration.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1522510-g011.tif"/>
</fig>
<p>The backend platform, based on our proposed algorithm, provides a comprehensive management interface, facilitating efficient, real-time collection of pest monitoring data from greenhouses and farms. Data can be flexibly submitted by agricultural technicians via smartphones or automatically uploaded by pest induction and laser capture devices. The backend management system automatically identifies pests, clearly visualizes real-time identification results, and assigns data to corresponding greenhouse or farmland regions according to geographic locations. The detailed system processing workflow is presented in the lower-right section of <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>.</p>
<p>To further validate the practical efficacy of our proposed few-shot pest insect detection model, we conducted an extensive field evaluation over a three-month period in vegetable greenhouses located in Haikou, Hainan Province. Situated in a tropical region, Hainan faces significant pest challenges. The evaluation specifically targeted eight prevalent pest species in this region: flea beetles, aphids, whiteflies, thrips, diamondback moths, armyworms, fruit&#xa0;flies, and leaf miners. We deployed a detection platform utilizing&#xa0;our proposed algorithm, continuously monitoring pest instances&#xa0;captured through smartphone images provided by agricultural&#xa0;technicians and integrated intelligent trapping devices. Throughout the evaluation period, a total of 563 pest instances were captured across all monitored areas. Among these, the AI model successfully identified 534 instances, yielding an overall accuracy of 94.84%. Notably, aphids and whiteflies demonstrated the highest detection accuracy, each exceeding 96%. In contrast, flea beetles exhibited slightly lower accuracy at 89.7% due to their smaller size and higher mobility.</p>
<p>Our methodology comprehensively addresses the dynamic and complex nature of pest monitoring environments by employing targeted detection strategies that integrate crop types, regional characteristics, and seasonal factors, significantly reducing data collection and labeling costs through few-shot learning techniques. The lightweight model design ensures effective deployment even in agricultural scenarios with limited computational resources or poor network connectivity, exhibiting robust and stable performance in greenhouse monitoring environments.</p>
</sec>
</sec>
</sec>
<sec id="s5" sec-type="conclusion">
<label>5</label>
<title>Conclusion</title>
<p>This study presents a novel FSOD method for pest insects, addressing challenges related to limited annotation data and multi object sizes. Built upon the Faster R-CNN framework, our approach integrates feature aggregation and SCL to enhance feature representation and improve detection accuracy. Multi-scale feature extraction using a Feature Pyramid Network captures rich semantic information at different scales, improving sensitivity to multi targets. A Feature Aggregation Module (FAM) with attention mechanism fuses features from the support and query sets, enhancing detection ability for small-sample targets. SCL is introduced to improve feature discriminability, while class weights and Focal Loss address class imbalance and hard-to-classify samples. Joint optimization of multiple tasks with an integrated loss function enhances robustness and precision. Experimental results demonstrate significant performance improvements in small and minority class pest detection, offering a valuable solution for agricultural pest management. While the proposed method achieves significant improvements in detection accuracy, the computational cost associated with Faster R-CNN remains a limitation for real-time applications. Future research could focus on optimizing the framework for faster inference or exploring lightweight architectures to enhance scalability for edge deployment.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>SH: Conceptualization, Formal analysis, Methodology, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. BJ: Data curation, Software, Writing &#x2013; original draft. XS: Visualization, Validation, Writing &#x2013; original draft. WJ: Project administration, Investigation, Validation, Writing &#x2013; review &amp; editing. JG: Writing &#x2013; review &amp; editing. FG: Methodology, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the Hainan Province Key R&amp;D Program Project (No. ZDYF2021GXJS010), the Major Science and Technology Project of Haikou City (No. 2020006), the Hainan Provincial Education and Teaching Reform Project of Colleges and Universities (No. Hnjg2021-37), and the Education and Teaching Reform Research Project of Hainan Province (No. Hnjg2019-50).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ali</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Dhanaraj</surname> <given-names>R. K.</given-names>
</name>
<name>
<surname>Kadry</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>AI-enabled IoT-based pest prevention and controlling system using sound analytics in large agricultural field</article-title>. <source>Comput. Electron. Agric.</source> <volume>220</volume>, <fpage>108844</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.108844</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>An</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Weng</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Insect recognition based on complementary features from multiple views</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>2966</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-023-29600-1</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anwar</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Masood</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Exploring deep ensemble model for insect and pest detection from images</article-title>. <source>Proc. Comput. Sci.</source> <volume>218</volume>, <fpage>2328</fpage>&#x2013;<lpage>2337</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.procs.2023.01.208</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arg&#xfc;eso</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Picon</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Irusta</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Medela</surname> <given-names>A.</given-names>
</name>
<name>
<surname>San-Emeterio</surname> <given-names>M. G.</given-names>
</name>
<name>
<surname>Bereciartua</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Few-Shot Learning approach for plant disease classification using images taken in the field</article-title>. <source>Comput. Electron. Agric.</source> <volume>175</volume>, <fpage>105542</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105542</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hou</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>A lightweight pest detection model for drones based on transformer and super-resolution sampling techniques</article-title>. <source>Agriculture</source> <volume>13</volume>, <fpage>1812</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture13091812</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Butera</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Ferrante</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Jermini</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Prevostini</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Alippi</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Precise agriculture: effective deep learning strategies to detect pest insects</article-title>. <source>IEEE/CAA. J. Autom. Sin.</source> <volume>9</volume>, <fpage>246</fpage>&#x2013;<lpage>258</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JAS.2021.1004317</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A sheep dynamic counting scheme based on the fusion between an improved-sparrow-search YOLOv5x-ECA model and few-shot deepsort algorithm</article-title>. <source>Comput. Electron. Agric.</source> <volume>206</volume>, <fpage>107696</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.107696</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Cui</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Meta-learning for few-shot plant disease detection</article-title>. <source>Foods</source> <volume>10</volume>, <fpage>2441</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/foods10102441</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Agricultural disease image dataset for disease identification based on machine learning</article-title>,&#x201d; in <conf-name>Big Scientific Data Management: First International Conference, BigSDM 2018</conf-name>, <conf-loc>Beijing, China</conf-loc>, <conf-date>November 30&#x2013;December 1, 2018</conf-date>. <fpage>263</fpage>&#x2013;<lpage>274</lpage> (<publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name>), Revised Selected Papers 1.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Egusquiza</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Picon</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Irusta</surname> <given-names>U.</given-names>
</name>
<name>
<surname>Bereciartua-Perez</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Eggers</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Klukas</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Analysis of few-shot techniques for fungal plant disease classification and evaluation of clustering capabilities over real datasets</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>813237</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.813237</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>A fast and lightweight detection model for wheat fusarium head blight spikes in natural environments</article-title>. <source>Comput. Electron. Agric.</source> <volume>216</volume>, <fpage>108484</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108484</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gomes</surname> <given-names>J. C.</given-names>
</name>
<name>
<surname>Borges</surname> <given-names>L. A. B.</given-names>
</name>
<name>
<surname>Borges</surname> <given-names>D. L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A multi-layer feature fusion method for few-shot image classification</article-title>. <source>Sensors</source> <volume>23</volume>, <fpage>6880</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s23156880</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>G. S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Few-shot object detection via variational feature aggregation</article-title>. <source>Proc. AAAI. Conf. Artif. Intell.</source> <volume>37</volume>, <fpage>755</fpage>&#x2013;<lpage>763</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1609/aaai.v37i1.25153</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Dou</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>He</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Boosting segment anything model towards open-vocabulary learning</article-title>. <source>arXiv. preprint. arXiv:2312.03628</source> <volume>39</volume> (<issue>3</issue>), <fpage>3356</fpage>&#x2013;<lpage>3365</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1609/aaai.v39i3.32347</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Xue</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A survey of deep learning for low-shot object detection</article-title>. <source>ACM Comput. Surveys.</source> <volume>56</volume>, <fpage>1</fpage>&#x2013;<lpage>37</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3570326</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khanam</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Yolov11: An overview of the key architectural enhancements</article-title>. <source>arXiv. preprint. arXiv:2410.17725</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2410.17725</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hua</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An effective object detector via diffused graphic large selective kernel with one-to-few labelling strategy for small-scaled crop diseases detection</article-title>. <source>Crop Prot.</source> <volume>182</volume>, <fpage>106705</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cropro.2024.106705</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chao</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Semi-supervised few-shot learning approach for plant diseases recognition</article-title>. <source>Plant Methods</source> <volume>17</volume>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-021-00770-1</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <source>A scheme for pest-dense area localization with solar insecticidal lamps Internet of Things under asymmetric links</source> (<publisher-loc>Piscataway, NJ, USA</publisher-loc>: <publisher-name>IEEE Transactions on AgriFood Electronics</publisher-name>).</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Demir</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Data-driven few-shot crop pest detection based on object pyramid for smart agriculture</article-title>. <source>J. Electron. Imaging</source> <volume>32</volume>, <fpage>052403</fpage>&#x2013;<lpage>052403</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1117/1.JEI.32.5.052403</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Few-shot cotton leaf spots disease classification based on metric learning</article-title>. <source>Plant Methods</source> <volume>17</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-021-00813-7</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Qiang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Tse</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>S. K.</given-names>
</name>
<name>
<surname>Pau</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A few-shot learning method for tobacco abnormality identification</article-title>. <source>Front. Plant Sci.</source> <volume>15</volume>, <elocation-id>1333236</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2024.1333236</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Tse</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>S. K.</given-names>
</name>
<name>
<surname>Qiang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Pau</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>a). <article-title>Few-shot learning for plant-disease recognition in the frequency domain</article-title>. <source>Plants</source> <volume>11</volume>, <fpage>2814</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/plants11212814</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Tse</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>S. K.</given-names>
</name>
<name>
<surname>Qiang</surname> <given-names>Z. P.</given-names>
</name>
<name>
<surname>Pau</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>b). <article-title>Few-shot learning approach with multi-scale feature fusion and attention for plant disease recognition</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>907916</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.907916</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhuo</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A dataset for forestry pest identification</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>857104</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.857104</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mahmood</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Matin</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Goudos</surname> <given-names>S. K.</given-names>
</name>
<name>
<surname>Karagiannidis</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Machine Learning for Smart Agriculture: A Comprehensive Survey</source> (<publisher-loc>Piscataway, NJ, USA</publisher-loc>: <publisher-name>IEEE Transactions on Artificial Intelligence</publisher-name>).</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A real-time object detection model for orchard pests based on improved YOLOv4 algorithm</article-title>. <source>Sci. Rep.</source> <volume>12</volume>, <fpage>13557</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-022-17826-4</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>P&#xf6;hler</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Eisenbach</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Gross</surname> <given-names>H. M.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Few-shot object detection: A comprehensive survey</source> (<publisher-loc>Piscataway, New Jersey, USA</publisher-loc>: <publisher-name>IEEE Transactions on Neural Networks and Learning Systems</publisher-name>).</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Popescu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Dinca</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ichim</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Angelescu</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>New trends in detection of harmful insects and pests in modern agriculture using artificial neural networks. a review</article-title>. <source>Front. Plant Sci.</source> <volume>14</volume>, <elocation-id>1268167</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2023.1268167</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ragu</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Teo</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Object detection and classification using few-shot learning in smart agriculture: A scoping mini review</article-title>. <source>Front. Sustain. Food Syst.</source> <volume>6</volume>, <elocation-id>1039299</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fsufs.2022.1039299</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rai</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>WeedVision: A single-stage deep learning architecture to perform weed detection and segmentation using drone-acquired images</article-title>. <source>Comput. Electron. Agric.</source> <volume>219</volume>, <fpage>108792</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.108792</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Faster R-CNN: Towards real-time object detection with region proposal networks</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>39</volume>, <fpage>1137</fpage>&#x2013;<lpage>1149</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2016.2577031</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rezaei</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Diepeveen</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Laga</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>M. G.</given-names>
</name>
<name>
<surname>Sohel</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Plant disease recognition in a low data scenario using few-shot learning</article-title>. <source>Comput. Electron. Agric.</source> <volume>219</volume>, <fpage>108812</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.108812</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>B.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>High-accuracy maize disease detection based on attention generative adversarial network and few-shot learning</article-title>. <source>Plants</source> <volume>12</volume>, <fpage>3105</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/plants12173105</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Fsce: Few-shot object detection via contrastive proposal encoding</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. (<publisher-loc>Piscataway, NJ, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>). <fpage>7352</fpage>&#x2013;<lpage>7362</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Teng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>MSR-RCNN: a multi-class crop pest detection network based on a multi-scale super-resolution feature enhancement module</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>810546</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.810546</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>uthalapati</surname> <given-names>S. V.</given-names>
</name>
<name>
<surname>Tunga</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Multi-domain few-shot learning and dataset for agricultural applications</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF International Conference on Computer Vision</conf-name>. (<publisher-loc>Piscataway, NJ, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1399</fpage>&#x2013;<lpage>1408</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Prior knowledge auxiliary for few-shot pest detection in the wild</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>1033544</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.1033544</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Grijalva</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Caragea</surname> <given-names>D.</given-names>
</name>
<name>
<surname>McCornack</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Detecting common coccinellids found in sorghum using deep learning models</article-title>. <source>Sci. Rep.</source> <volume>13</volume>, <fpage>9748</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-023-36738-5</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>T. E.</given-names>
</name>
<name>
<surname>Darrel</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Gonzalez</surname> <given-names>J. E.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Frustratingly simple few-shot object detection</article-title>,&#x201d; in <conf-name>Proceedings of the Conference on International Conference on Machine Learning</conf-name>, <conf-loc>Vienna, Austria. New York</conf-loc>. <fpage>9919</fpage>&#x2013;<lpage>9928</lpage> (<publisher-name>ACM</publisher-name>).</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Teng</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Few-shot vegetable disease recognition model based on image text collaborative representation learning</article-title>. <source>Comput. Electron. Agric.</source> <volume>184</volume>, <fpage>106098</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2021.106098</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Su</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Pest-YOLO: A model for large-scale multi-class dense and tiny pest detection and counting</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>, <elocation-id>973985</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.973985</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhan</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>Y. K.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J. l.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Ip102: A large-scale benchmark dataset for insect pest recognition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. (<publisher-loc>Piscataway, NJ, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>8787</fpage>&#x2013;<lpage>8796</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Marinello</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Ercisli</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A survey of few-shot learning in smart agriculture: developments, applications, and challenges</article-title>. <source>Plant Methods</source> <volume>18</volume>, <fpage>28</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-022-00866-2</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Image-fusion-based object detection using a time-of-flight camera</article-title>. <source>Optics. Express.</source> <volume>31</volume>, <fpage>43100</fpage>&#x2013;<lpage>43114</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1364/OE.510101</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>SRNet-YOLO: A model for detecting tiny and very tiny pests in cotton fields based on super-resolution reconstruction</article-title>. <source>Front. Plant Sci.</source> <volume>15</volume>, <elocation-id>1416940</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2024.1416940</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Cui</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>A comprehensive survey on segment anything model for vision and beyond</article-title>. <source>arXiv. preprint. arXiv:2305.08196</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2305.08196</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Zero-and few-shot learning for diseases recognition of Citrus aurantium L. using conditional adversarial autoencoders</article-title>. <source>Comput. Electron. Agric.</source> <volume>179</volume>, <fpage>105828</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2020.105828</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Leveraging the feature distribution calibration and data augmentation for few-shot classification in fish counting</article-title>. <source>Comput. Electron. Agric.</source> <volume>212</volume>, <fpage>108151</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108151</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>