<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Environ. Sci.</journal-id>
<journal-title>Frontiers in Environmental Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Environ. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-665X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1391770</article-id>
<article-id pub-id-type="doi">10.3389/fenvs.2024.1391770</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Environmental Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A pest image recognition method for long-tail distribution problem</article-title>
<alt-title alt-title-type="left-running-head">Chen et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fenvs.2024.1391770">10.3389/fenvs.2024.1391770</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Shengbo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2575734/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gao</surname>
<given-names>Quan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2726926/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>He</surname>
<given-names>Yun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2412864/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>College of Big Data</institution>, <institution>Yunnan Agricultural University</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Key Laboratory for Crop Production and Intelligent Agriculture of Yunnan Province</institution>, <addr-line>Kunming</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/893641/overview">Sushant K. Singh</ext-link>, CAIES Foundation, India</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/895350/overview">Loris Nanni</ext-link>, University of Padua, Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1042530/overview">Guofeng Yang</ext-link>, Zhejiang University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yun He, <email>heyun@ynau.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>29</day>
<month>07</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>12</volume>
<elocation-id>1391770</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>02</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Chen, Gao and He.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Chen, Gao and He</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Deep learning has revolutionized numerous fields, notably image classification. However, conventional methods in agricultural pest recognition struggle with the long-tail distribution of pest image data, characterized by limited samples in rare pest categories, thereby impeding overall model performance. This study proposes two state-of-the-art techniques: Instance-based Data Augmentation (IDA) and Constraint-based Feature Tuning (CFT). IDA collaboratively applies resampling and mixup methods to notably enhance feature extraction for rare class images. This approach addresses the long-tail distribution challenge through resampling, ensuring adequate representation for scarce categories. Additionally, by introducing data augmentation, we further refined the recognition of tail-end categories without compromising performance on common samples. CFT, a refinement built upon pre-trained models using IDA, facilitated the precise classification of image features through fine-tuning. Our experimental findings validate that our proposed method outperformed previous approaches on the CIFAR-10-LT, CIFAR-100-LT, and IP102 datasets, demonstrating its effectiveness. Using IDA and CFT to optimize the ViT model, we observed significant improvements over the baseline, with accuracy rates reaching 98.21%, 88.62%, and 64.26%, representing increases of 0.74%, 3.55%, and 5.73% respectively. Our evaluation of the CIFAR-10-LT and CIFAR-100-LT datasets also demonstrated state-of-the-art performance.</p>
</abstract>
<kwd-group>
<kwd>insect pest recognition</kwd>
<kwd>long-tail distribution data</kwd>
<kwd>data augmentation</kwd>
<kwd>deep machine learning</kwd>
<kwd>classification</kwd>
</kwd-group>
<contract-num rid="cn001">32101611</contract-num>
<contract-num rid="cn002">202302AE090020 202202AE090021</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Major Science and Technology Projects in Yunnan Province<named-content content-type="fundref-id">10.13039/501100018531</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Big Data, AI, and the Environment</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Crop production is significantly influenced by factors including the availability of water, soil quality, light, temperature, and climatic conditions (Ren et al., 2019). Additionally, pests pose a considerable threat, causing diseases, leaf and fruit damage, and overall yield reduction. Specifically, among crops, the total global potential loss due to pests varied from about 50% in wheat to more than 80% in cotton production. The responses are estimated as losses of 26%&#x2013;29% for soybean, wheat, and cotton, and 31%, 37%, and 40% for maize, rice, and potatoes (<xref ref-type="bibr" rid="B37">Oerke, 2006</xref>). Consequently, effective pest management is imperative for crop health and productivity optimization (<xref ref-type="bibr" rid="B59">Wu et al., 2019</xref>). Traditional pest identification methods suffer from drawbacks such as subjective human observation, inconsistent results, reliance on extensive reference materials, and the potential oversight of small or concealed pests (<xref ref-type="bibr" rid="B30">Liu et al., 2020</xref>). These limitations impede accuracy, require labor-intensive efforts, and hinder efficient identification. To address these problems, there is a crucial need for a more efficient approach. Advanced approaches have also been proposed in the agricultural domain, such as ResNet variants (<xref ref-type="bibr" rid="B15">Dewi et al., 2023</xref>; <xref ref-type="bibr" rid="B63">Zhang et al., 2023</xref>) and data augmentation (<xref ref-type="bibr" rid="B38">Patel and Bhatt, 2021</xref>; <xref ref-type="bibr" rid="B41">Qian et al., 2023</xref>). And several quantitative techniques have been explored for data-driven decision-making in pest and disease recognition. <xref ref-type="bibr" rid="B53">Spinelli et al. (2004)</xref> assessed a near infrared (NIR)-based technique for detecting fire blight disease in asymptomatic pear plants under greenhouse conditions, utilizing quantitative NIR spectroscopy to measure spectral reflectance and identify disease presence before visible symptoms appeared. <xref ref-type="bibr" rid="B21">Huang and Apan (2006)</xref> collected hyperspectral data using a portable spectrometer under field conditions to detect Sclerotinia rot disease in celery, employing hyperspectral imaging for precise analysis of spectral signatures to differentiate diseased and healthy tissues. <xref ref-type="bibr" rid="B50">Shafri and Hamdan (2009)</xref> used airborne hyperspectral imaging to detect ganoderma basal stem rot disease in oil palm plantations, providing high-resolution, quantitative data on spectral properties, which enabled early and accurate disease detection through detailed spectral analysis. These researchers have developed methods to address the recognition of specific scenarios or certain crop pests and diseases, achieving high accuracy and automation in pest identification. However, due to the prevalent long-tail distribution phenomenon in pest image data, applying these models often results in inadequate generalization ability and low recognition accuracy. Therefore, based on simple pre-trained models, in long-tail distribution datasets, our methods enhance the model&#x2019;s performance by ensuring adequate representation for scarce categories and improving feature extraction and classification accuracy.</p>
<p>In recent years, the advent of large-scale labeled datasets, increased computational power, and innovations in algorithms and architectures have enabled deep learning to excel in automated feature extraction and high accuracy, making it widely applicable in image recognition. Several researchers have introduced deep learning into pest recognition, addressing specific challenges in the field. For instance, <xref ref-type="bibr" rid="B46">Samanta and Ghosh (2012)</xref> applied neural networks with CFS for tea pest classification, achieving perfect accuracy. <xref ref-type="bibr" rid="B45">Salih et al. (2020)</xref> used CNNs for accurate tomato disease classification with deep learning. <xref ref-type="bibr" rid="B49">Sethy et al. (2020)</xref> applied CNNs for rice disease classification, outperforming traditional methods with deep learning. <xref ref-type="bibr" rid="B63">Zhang et al. (2023)</xref> and <xref ref-type="bibr" rid="B15">Dewi et al. (2023)</xref> have optimized ResNet architectures to address issues encountered in pest recognition. Image classification allows computers to automatically understand features within an image, identify objects or scenes depicted within it, and assign them to appropriate categories. Researchers, such as <xref ref-type="bibr" rid="B5">Coulibaly et al. (2022a)</xref>, have proposed deep convolutional neural networks based on CNN algorithms for insect pest recognition. In (<xref ref-type="bibr" rid="B44">Ren et al., 2019</xref>; <xref ref-type="bibr" rid="B30">Liu et al., 2020</xref>), ResNet variants were designed for insect pest recognition, integrating explainability features and demonstrating exceptional performance on certain datasets. In their work, this process typically includes several steps: data collection, data preprocessing, model construction, feature extraction, model training, model evaluation, hyperparameter tuning, and model optimization. However, during pest image data collection, there is often a disparity in the number of samples, with some pests being abundant and others scarce, leading to a long-tail distribution in the dataset. To address this, researchers often employ data augmentation techniques such as rotation and cropping during preprocessing. However, these methods only increase the quantity of existing samples without fundamentally changing the original dataset. Although these methods improve model accuracy on long-tail datasets to some extent, their effectiveness is limited. Moreover, these researchers tend to focus on optimizing models to solve the pest recognition problem. When applying these optimized models to long-tail distribution datasets, the models often do not perform as well as expected. To achieve better recognition accuracy on long-tail distribution datasets, this paper introduces an integrated data augmentation technique that combines resampling, self-attention mechanisms, and hybrid methods. Additionally, a phased approach to training models is proposed.</p>
<p>When handling images with a long-tailed distribution, models often struggle to classify the tail-end categories, yielding suboptimal fits. Although correcting feature representation methods can enhance model performance, the effectiveness of such methods appears limited (<xref ref-type="bibr" rid="B60">Yang et al., 2021</xref>; <xref ref-type="bibr" rid="B58">Wang et al., 2023</xref>). Therefore, an alternative approach should enhance the model&#x2019;s classification accuracy and generalization capabilities by modifying the mapping correlation between image features and their corresponding classes post-feature correction.</p>
<p>This study addresses the challenges posed by the long-tailed distribution of pest images by introducing Instance-Based Data Augmentation (IDA) and Constraint-Based Feature Tuning (CFT). IDA exhibits resemblances to Mixup (<xref ref-type="bibr" rid="B62">Zhang et al., 2018</xref>). While Mixup enhances classification performance, the generated interpolated samples often lack naturalness. Mixup does not address long-tailed sample distributions, limiting its ability to improve minority class representations.</p>
<p>On the contrary, IDA addresses this issue by incorporating resampling and self-attention mechanisms. This approach employs resampling to address data scarcity, particularly for underrepresented tail-end categories. This rebalancing strategy effectively mitigates bias caused by an uneven sample distribution, resulting in a more equitable and representative dataset. Additionally, integrating self-attention mechanisms enables the model to discern intricate relationships among samples. Moreover, integrating self-attention mechanisms allows the model to capture the underlying data structures, enhancing classification performance across all categories. Consequently, IDA fine-tunes classification results and alleviates the generation of unnatural interpolated samples. CFT endeavors restrict the adjustment of model parameters post-feature extraction. Selective optimization of a limited subset of parameters can bolster image feature classification, elevate the model&#x2019;s generalization prowess, and prevent overfitting.</p>
<p>To achieve an efficient pest classifier, we trained the model on the IP102 dataset (<xref ref-type="bibr" rid="B59">Wu et al., 2019</xref>), a large-scale benchmark dataset for insect pest recognition with a natural long-tailed distribution. To alleviate the data imbalance issue, a weighted loss function has been employed to address imbalanced learning among various types (<xref ref-type="bibr" rid="B29">Lin et al., 2017</xref>; <xref ref-type="bibr" rid="B28">Li et al., 2020</xref>). Researchers have explored various techniques such as resampling, adversarial augmentation, and ensemble learning.</p>
<p>The contributions of this study are summarized as follows.<list list-type="simple">
<list-item>
<p>1. We refined the feature mapping process within large-scale neural networks to neutralize the adverse effects of data imbalance. This advancement strengthened the network&#x2019;s capability to learn from and recognize underrepresented categories.</p>
</list-item>
<list-item>
<p>2. We introduced an IDA technique integrating resampling, self-attention mechanisms, and mixup approaches. This comprehensive method effectively tackles class imbalance, diversifies datasets, and augments overall model performance.</p>
</list-item>
<list-item>
<p>3. We proposed CFT, optimizing the MLP parameters while locking others. This optimization improved image feature classification by focusing on enriching MLP performance.</p>
</list-item>
<list-item>
<p>4. The effectiveness of our proposed method is validated on the IP102 dataset, attaining state-of-the-art performance in pest identification within this dataset.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2">
<title>2 Literature review</title>
<sec id="s2-1">
<title>2.1 Insect pest recognition</title>
<p>Insect pests pose significant threats to crop yields, making early pest identification crucial for maximizing the quality and yield of agricultural products to avert economic losses. Insect pest recognition methods can be categorized into handcrafted and deep learning techniques.</p>
<p>Handcrafted methodologies for insect pest recognition, such as SIFT (<xref ref-type="bibr" rid="B33">Lowe, 2004</xref>) and HOG (<xref ref-type="bibr" rid="B11">Dalal and Triggs, 2005</xref>), have been widely utilized for insect pest identification (<xref ref-type="bibr" rid="B46">Samanta and Ghosh, 2012</xref>; <xref ref-type="bibr" rid="B43">Rani and Amsini, 2016</xref>). Although these methods are effective, they come with their own set of limitations. SIFT struggles with scale variations, while HOG&#x2019;s performance is hampered by images with complex backgrounds, leading to their replacement by deep learning approaches to handle diverse image conditions.</p>
<p>The advancement of deep learning, particularly CNNs, has revolutionized various fields including plant diseases identification (<xref ref-type="bibr" rid="B56">Vaswani et al., 2017</xref>; <xref ref-type="bibr" rid="B45">Salih et al., 2020</xref>; <xref ref-type="bibr" rid="B49">Sethy et al., 2020</xref>), plant recognition (<xref ref-type="bibr" rid="B18">Dyrmann et al., 2016</xref>), and insect pest recognition (Ren et al., 2019; <xref ref-type="bibr" rid="B59">Wu et al., 2019</xref>). CNNs are crucial for insect classification tasks. However, existing methods are insufficient for accurately detecting rice pests with variable shapes or similar appearances. To address this issue, <xref ref-type="bibr" rid="B26">Li S. et al. (2022)</xref> proposed a self-attention feature fusion model for rice pest detection (SAFFPest), significantly improving the identification compared to previous methods. Furthermore, several CNN variants have been developed for insect pest recognition. For instance, <xref ref-type="bibr" rid="B55">Ung et al. (2021)</xref> investigated various CNN-based architectures, integrating attention mechanisms, feature pyramid networks, and fine-grained models. <xref ref-type="bibr" rid="B36">Nanni et al. (2022)</xref> proposed a technique for insect classification using CNNs along with innovative variants of the Adam optimization algorithm. <xref ref-type="bibr" rid="B6">Coulibaly et al. (2022b)</xref> introduced a CNN-based method for identifying and localizing insect pests by integrating techniques for enhancing model interpretability, leveraging visualization maps to highlight key color and shape features captured by the CNNs. The above-mentioned strategies yielded promising outcomes across diverse large-scale pest-related datasets, demonstrating improvements through adjustments to neural network architectures and the integration of novel modules. However, in practical scenarios, pests display natural long-tail distributions rather than conforming to pre-recognition based on relatively balanced datasets. <xref ref-type="bibr" rid="B59">Wu et al. (2019)</xref> assembled a substantial dataset named IP102 for insect pest recognition, comprising over 75,000 images distributed across 102 categories, characterized by a natural long-tailed distribution, as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Distribution of categories and quantities in IP102.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g001.tif"/>
</fig>
<p>Meanwhile, models such as Alexnet (<xref ref-type="bibr" rid="B24">Krizhevsky et al., 2012</xref>), GoogleNet (<xref ref-type="bibr" rid="B54">Szegedy et al., 2015</xref>), VGGNet (<xref ref-type="bibr" rid="B52">Simonyan and Zisserman, 2015</xref>), and ResNet (<xref ref-type="bibr" rid="B20">He et al., 2015</xref>) have been utilized on IP102 dataset, albeit with suboptimal performances in respective domains. <xref ref-type="bibr" rid="B27">Li W. et al. (2022)</xref> addressed this issue by employing advanced deep learning techniques to train YOLOv5 and Faster-RCNN ResNet50 on the IP102, mitigating low detection and recognition accuracy, particularly in scenarios of detecting multiple complex sample types. Ren et al. (2019) proposed a Feature Reuse Residual Network (FR-ResNet) and assessed its efficacy on the IP102 reference dataset. Experimental findings revealed that FR-ResNet could achieve favorable performance improvement in insect pest classification. <xref ref-type="bibr" rid="B60">Yang et al. (2021)</xref> devised a Convolutional Rebalancing Network to classify rice pests and diseases using field image datasets, enhancing classification performance on long-tailed datasets. <xref ref-type="bibr" rid="B58">Wang et al. (2023)</xref> introduced a deep learning architecture that combined ConvNeXt and Swin Transformer models to address classification challenges in long-tailed pest datasets, surpassing the performance of existing methods. Most current approaches aim to improve the recognition accuracy of long-tailed distribution datasets, and there is no effective technique based on data augmentation and optimization of feature concealment.</p>
</sec>
<sec id="s2-2">
<title>2.2 Data augmentation for image classification</title>
<p>Data augmentation has emerged as a pivotal strategy for bolstering the performance of machine learning models, particularly in scenarios where training data is scarce. Several data augmentation techniques have been introduced to artificially broaden the dataset&#x2019;s variability and enhance the model&#x2019;s overall generalization capabilities.</p>
<p>For image-based augmentation, various techniques such as rotation, flipping, scaling, and cropping (<xref ref-type="bibr" rid="B51">Simard et al., 2003</xref>; <xref ref-type="bibr" rid="B57">Wan et al., 2013</xref>; <xref ref-type="bibr" rid="B48">Sato et al., 2015</xref>) have been extensively utilized to generate modified versions of original images. Horizontal and vertical flips simulate different perspectives, while rotations mimic changes in object orientation. These transformations effectively expand the dataset, enabling the model to better handle variations encountered in real-world scenarios. Regarding color augmentation, adjusting brightness, contrast, saturation (<xref ref-type="bibr" rid="B24">Krizhevsky et al., 2012</xref>), and hue to modify the color characteristics of images has proven beneficial. These adjustments mimic diverse lighting conditions and contribute to a more comprehensive understanding of the data.</p>
<p>In recent years, cutout (<xref ref-type="bibr" rid="B14">Devries and Taylor, 2017</xref>) and cutmix (<xref ref-type="bibr" rid="B61">Yun et al., 2019</xref>) have significantly advanced data augmentation techniques. Cutout involves masking random patches from images, prompting the model to focus on other relevant features. CutMix blends portions of different images, compelling the model to learn from mixed information. These methods promote better generalization of the model while mitigating overfitting. To enhance data augmentation strategies, studies (<xref ref-type="bibr" rid="B7">Cubuk et al., 2019</xref>; <xref ref-type="bibr" rid="B8">2020</xref>) have employed the search algorithm to determine the optimal policy and used a smaller proxy task to overcome the expense of the search phase. Generative adversarial networks (<xref ref-type="bibr" rid="B19">Goodfellow et al., 2014</xref>) are also used to generate additional effective data as data augmentation (<xref ref-type="bibr" rid="B1">Antoniou et al., 2017</xref>; <xref ref-type="bibr" rid="B35">Mun et al., 2017</xref>; <xref ref-type="bibr" rid="B39">Perez and Wang, 2017</xref>; <xref ref-type="bibr" rid="B67">Zhu et al., 2017</xref>).</p>
<p>In the field of pest control, researchers have also adopted data augmentation techniques to enhance their models&#x2019; performance. For instance, <xref ref-type="bibr" rid="B25">Kusrini et al. (2020)</xref> implemented mathematical operations to images, exploring various combinations of these operations. <xref ref-type="bibr" rid="B38">Patel and Bhatt (2021)</xref> tackled class imbalance by incorporating augmentation parameters such as horizontal flip and 90-degree rotation. Additionally, <xref ref-type="bibr" rid="B41">Qian et al. (2023)</xref> introduced an innovative automatic data augmentation method to dynamically search for suitable augmentation strategies. These studies utilized data augmentation to improve the accuracy of pest identification. However, their applications were limited to datasets of specific types of pests or diseases.</p>
<p>While these methods contribute to general data enhancement, pest control presents unique challenges due to its sensitivity to biological features (<xref ref-type="bibr" rid="B49">Sethy et al., 2020</xref>), adaptability to environmental changes (<xref ref-type="bibr" rid="B59">Wu et al., 2019</xref>), and a natural long-tail pattern distribution. Unlike conventional computer vision tasks that emphasize shapes and textures, pest identification considers insect physiology, pose variations, and appearance changes in different growth stages. This distinctiveness underscores the necessity for a deeper understanding of insect biology to devise effective data augmentation. Generic computer vision methods may not be optimal in this context, highlighting the importance of our proposed research.</p>
</sec>
<sec id="s2-3">
<title>2.3 Pre-training and fine-tuning</title>
<p>A pre-trained model is utilized due to its capacity to extract both fundamental and complex features during training on extensive datasets (<xref ref-type="bibr" rid="B40">Poth et al., 2021</xref>). This versatility renders pre-trained models invaluable for transfer learning, furnishing a foundation of adaptable features across various domains.</p>
<p>When the dataset aligns with the pre-trained model&#x2019;s dataset, fine-tuning emerges as a viable strategy, particularly in transfer learning. Fine-tuning finds broad application in natural language processing and computer vision. Compared to training a model from scratch, fine-tuning offers an intuitive solution and yields substantial enhancements across computational efficiency, training duration, model efficacy, and precision. In natural language processing, researchers often leverage pre-trained language models such as BERT (<xref ref-type="bibr" rid="B13">Devlin et al., 2019</xref>) or GPT (<xref ref-type="bibr" rid="B42">Radford et al., 2018</xref>). These models, trained on extensive text corpora, capture nuanced representations of language structures and contexts. Through fine-tuning, model parameters can be tailored to suit diverse tasks such as text classification (<xref ref-type="bibr" rid="B64">Zheng et al., 2020a</xref>), sentiment analysis (<xref ref-type="bibr" rid="B2">Bataa and Wu, 2019</xref>), or named entity recognition (<xref ref-type="bibr" rid="B32">Liu et al., 2021</xref>), catering to the specific requirements across different domains.</p>
<p>In computer vision, CNNs excel as image feature extractors through pre-training on large-scale image classification tasks. Fine-tuning is commonly employed for tasks such as object detection (<xref ref-type="bibr" rid="B10">Dai et al., 2021</xref>) and image segmentation (<xref ref-type="bibr" rid="B3">Chaitanya et al., 2020</xref>), adapting the model to specific target tasks and achieving more precise image recognition and understanding.</p>
<p>In pest recognition, both pre-training and fine-tuning are pivotal. Transfer learning was applied by <xref ref-type="bibr" rid="B22">Kasinathan and Reddy (2019)</xref> to fine-tune pre-trained models, facilitating efficient classification of insect types in major crops. Additionally, <xref ref-type="bibr" rid="B31">Liu et al. (2022)</xref> combined transfer learning and fine-tuning to devise two transfer strategies within CNN for pest identification, significantly enhancing classification performance and effectively managing forest pests.</p>
<p>However, earlier studies predominantly relied on datasets with fewer categories or balanced distributions, limited to specific scenarios or regions, hindering widespread real-world applicability. The developed models failed to yield optimal results with real long-tail distributions. Given the limitations, our research focuses on enhancing the universal applicability of pest detection in real-life scenarios.</p>
</sec>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methodology</title>
<p>This section introduces the pre-trained model utilized for pest identification and network architecture. Then, the IDA approach employed with long-tailed distribution of image data is elucidated. Subsequently, we discuss the CFT technique. Finally, a detailed description of the loss function employed during the network training is provided. In this manuscript, the symbol &#x201c;I&#x201d; symbolizes the input image. The flowchart of method is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>In Stage 1, we implemented the IDA method for data augmentation, followed by the construction and training of a pre-trained model. In Stage 2, we employed CFT to adjust the correct mapping of features to categories. Post-data augmentation using the IDA method, we fixed most parameters in the pre-trained model and solely trained the MLP layer to achieve the mapping from features to categories.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g002.tif"/>
</fig>
<sec id="s3-1">
<title>3.1 Build pre-trained vision encoder</title>
<p>In recent years, the adoption of pre-trained vision models has surged within the field of computer vision. Leveraging well-established architectures such as ResNet and ViT (<xref ref-type="bibr" rid="B16">Dosovitskiy et al., 2020</xref>), pre-trained models have become pivotal in developing robust and high-performing solutions for various visual tasks. Initially trained on large-scale datasets, these models excel in capturing complex image features and patterns, leading to significant improvements in model generalization and performance on downstream tasks. Pre-trained models consistently outperform models trained from scratch due the former&#x2019;s adept feature extraction capabilities refined during pre-training. Pre-trained models inherently grasp fundamental visual concepts, including edges, textures, and shapes, contributing to their robustness across various tasks.</p>
<p>ResNet and ViT are two prominent architectures of pre-trained models. ResNet&#x2019;s deep structure, characterized by residual connections, effectively addresses the vanishing gradient problem, making the model a versatile feature extractor. Conversely, ViT utilizes attention mechanisms to process images by breaking them into smaller patches and flattening them for transformer-based processing, showcasing significant potential in the visual domain. Drawing upon the strengths of pre-trained models, the research community often relies on ResNet and ViT as foundational components for various visual tasks. Fine-tuning these models expedite the development of task-specific models. Therefore, by harnessing the knowledge encoded within pre-trained models, particularly ResNet and ViT, diverse visual challenges can be addressed, yielding enhanced performance and efficiency.</p>
<p>Therefore, we adopted these two models for feature extraction from the images in Eqs (<xref ref-type="disp-formula" rid="e1">1</xref>) and (<xref ref-type="disp-formula" rid="e2">2</xref>).<disp-formula id="e1">
<mml:math id="m1">
<mml:mi>f</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">s</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi mathvariant="normal">e</mml:mi>
<mml:mi mathvariant="normal">t</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mi>f</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf1">
<mml:math id="m3">
<mml:mi>f</mml:mi>
</mml:math>
</inline-formula> represents the image features acquired after the model&#x2019;s extraction process of ResNet or ViT with predetermined parameter set <inline-formula id="inf2">
<mml:math id="m4">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> that influence the model&#x2019;s internal operations; <inline-formula id="inf3">
<mml:math id="m5">
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> signifies a Residual Neural Network, a specialized deep learning architecture; <inline-formula id="inf4">
<mml:math id="m6">
<mml:mi>V</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>T</mml:mi>
</mml:math>
</inline-formula> denotes the Vision Transformer model, a prominent architecture for computer vision tasks; <inline-formula id="inf5">
<mml:math id="m7">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> represents the model parameters of the Multi-Layer Perceptron (MLP). We calculate the probability distribution using Eq. (<xref ref-type="disp-formula" rid="e3">3</xref>).<disp-formula id="e3">
<mml:math id="m8">
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf6">
<mml:math id="m9">
<mml:mi>p</mml:mi>
</mml:math>
</inline-formula> denotes the probability distribution, comprising the predicted probabilities, from the model for each possible class. These probabilities indicate the model&#x2019;s confidence in assigning the input to different classes. Finally, we employed the argmax function to compute the argument at which a function yields its maximum value, as shown in Eq. (<xref ref-type="disp-formula" rid="e4">4</xref>).<disp-formula id="e4">
<mml:math id="m10">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">a</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">g</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mi mathvariant="normal">x</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(4)</label>
</disp-formula>
<inline-formula id="inf7">
<mml:math id="m11">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> denotes the predicted class label in the final outcome. This methodology aims to enhance classification accuracy.</p>
</sec>
<sec id="s3-2">
<title>3.2 IDA</title>
<p>This study explored three approaches: class rebalancing, information augmentation, and model enhancement, to bolster model performance on long-tailed distribution data.</p>
<p>
<bold>Input:</bold> Graph dataset <inline-formula id="inf8">
<mml:math id="m12">
<mml:mi mathvariant="script">D</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, batch size <inline-formula id="inf9">
<mml:math id="m13">
<mml:mi>N</mml:mi>
</mml:math>
</inline-formula>, number of classes <inline-formula id="inf10">
<mml:math id="m14">
<mml:mi>c</mml:mi>
</mml:math>
</inline-formula>
</p>
<p>
<bold>Output:</bold> Augmented dataset <inline-formula id="inf11">
<mml:math id="m15">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>aug</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>
</p>
<p>
<bold>1</bold> <inline-formula id="inf12">
<mml:math id="m16">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>aug</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2190;</mml:mo>
<mml:mi mathvariant="script">D</mml:mi>
</mml:math>
</inline-formula>
</p>
<p>
<bold>2</bold> Class weights <inline-formula id="inf13">
<mml:math id="m17">
<mml:mi>w</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<p>
<bold>3</bold> Sampling weights <inline-formula id="inf14">
<mml:math id="m18">
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mi>w</mml:mi>
</mml:math>
</inline-formula>
</p>
<p>
<bold>4 foreach</bold> <italic>epoch</italic> <bold>do</bold>
</p>
<p>
<bold>5</bold> &#x2003;&#x2003;Sample mini-batch <inline-formula id="inf15">
<mml:math id="m19">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:math>
</inline-formula> from <inline-formula id="inf16">
<mml:math id="m20">
<mml:mi mathvariant="script">D</mml:mi>
</mml:math>
</inline-formula> based on sampling weights <inline-formula id="inf17">
<mml:math id="m21">
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>
</p>
<p>
<bold>6</bold> &#x2003;&#x2003;&#x2006;Sample pairs of indices <inline-formula id="inf18">
<mml:math id="m22">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> from <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<p>
<bold>7 &#x2003;&#x2003;&#x2006;foreach</bold> <inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">G</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> in <inline-formula id="inf21">
<mml:math id="m25">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> <bold>do</bold>
</p>
<p>
<bold>8</bold> &#x2003;&#x2003;&#x2003;&#x2006;Generate random mixing coefficients <inline-formula id="inf22">
<mml:math id="m26">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> with Eq. <xref ref-type="disp-formula" rid="e7">7</xref>
</p>
<p>
<bold>9</bold> &#x2003;&#x2003;&#x2003;&#x2006;Obtain new sample with Eq. <xref ref-type="disp-formula" rid="e5">5</xref>
</p>
<p>
<bold>10</bold> &#x2003;&#x2003;&#x2003;Generate new label with Eq. <xref ref-type="disp-formula" rid="e6">6</xref>
</p>
<p>
<bold>11</bold> &#x2003;&#x2003;&#x2003;Add new sample to <inline-formula id="inf23">
<mml:math id="m27">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>aug</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>
</p>
<p>
<bold>12 &#x2003;&#x2003;end</bold>
</p>
<p>
<bold>13 end</bold>
</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1. IDA.</label>
<p>&#x2003;</p>
</statement>
</p>
<p>The IDA technique employs data resampling strategies to balance category distribution and the Mixup approach, as outlined in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>. We utilized the Beta distribution to determine a mixing coefficient <inline-formula id="inf24">
<mml:math id="m28">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula>. When <inline-formula id="inf25">
<mml:math id="m29">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula> approaches 0, the resulting feature <inline-formula id="inf26">
<mml:math id="m30">
<mml:msup>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> retains a larger proportion of <inline-formula id="inf27">
<mml:math id="m31">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula>, while for <inline-formula id="inf28">
<mml:math id="m32">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula> close to 1, <inline-formula id="inf29">
<mml:math id="m33">
<mml:msup>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> bears a stronger resemblance to <inline-formula id="inf30">
<mml:math id="m34">
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>:<disp-formula id="e5">
<mml:math id="m35">
<mml:msup>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(5)</label>
</disp-formula>
<disp-formula id="e6">
<mml:math id="m36">
<mml:msup>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>Where <inline-formula id="inf31">
<mml:math id="m37">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf32">
<mml:math id="m38">
<mml:msup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> signify the original and modified image data, respectively, where the latter is obtained by shuffling; <inline-formula id="inf33">
<mml:math id="m39">
<mml:mi>l</mml:mi>
</mml:math>
</inline-formula> and <inline-formula id="inf34">
<mml:math id="m40">
<mml:msup>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> denote the original and modified labels of the original and modified images, respectively; <inline-formula id="inf35">
<mml:math id="m41">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, for <inline-formula id="inf36">
<mml:math id="m42">
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e7">
<mml:math id="m43">
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mspace width="0.3333em" class="nbsp"/>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x222b;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
</mml:math>
<label>(7)</label>
</disp-formula>When <inline-formula id="inf37">
<mml:math id="m44">
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula>, the Beta distribution degenerates into a uniform distribution.</p>
<p>IDA augments the information content of tail-end categories and overcomes sample diversity constraints, enhancing model generalization and addressing challenges in long-tailed distribution data.</p>
</sec>
<sec id="s3-3">
<title>3.3 CFT</title>
<p>The images&#x2019; data features become mixed, causing category feature mismatches, making them difficult to differentiate. In <xref ref-type="fig" rid="F3">Figure 3</xref> blending two sampled images creates new image features and labels. However, the classifier primarily relies on the blended features being from the original images rather than the new labels. Despite utilizing ResNet and ViT models, efficient mapping data features to categories was not achieved. Our approach in CFA recognized the complexity of disentangling spatial data features within images and prioritized optimizing the mapping relationship between data features and categories.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Augmented images obtained through IDA when <inline-formula id="inf38">
<mml:math id="m45">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula> &#x3d; 0.5.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g003.tif"/>
</fig>
<p>To achieve this goal, we adopted a two-stage process. In the first stage, two pre-trained models were employed to augment the model&#x2019;s knowledge and facilitate image feature extraction. Following training, the best-performing model was selected for the second stage. In the second stage, we fixed most parameters of this best-performing model, focusing on enhancing and fine-tuning the MLP layer. The model was then retrained to efficiently map extracted features to their respective categories.</p>
<p>We employed the L2 regularization, and through Eq. (<xref ref-type="disp-formula" rid="e8">8</xref>) updated <inline-formula id="inf39">
<mml:math id="m46">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, minimizing the loss function, denoted as &#x2018;Loss&#x2019; (<xref ref-type="sec" rid="s3-4">Section 3.4</xref>).<disp-formula id="e8">
<mml:math id="m47">
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-4">
<title>3.4 Loss function</title>
<p>In tasks involving multi-class classification, cross-entropy (<xref ref-type="bibr" rid="B12">De Boer et al., 2005</xref>) is frequently employed, as shown in Eq. (<xref ref-type="disp-formula" rid="e9">9</xref>).<disp-formula id="e9">
<mml:math id="m48">
<mml:mi>H</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>P</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>log</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(9)</label>
</disp-formula>where <inline-formula id="inf40">
<mml:math id="m49">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> indexes the various classes or outcomes; <inline-formula id="inf41">
<mml:math id="m50">
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf42">
<mml:math id="m51">
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> signify the true and predicted probability distributions, respectively.</p>
</sec>
<sec id="s3-5">
<title>3.4.1 First stage: optimizing entire model parameters</title>
<p>
<disp-formula id="e10">
<mml:math id="m52">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">stg1</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(10)</label>
</disp-formula>We used Eq. (<xref ref-type="disp-formula" rid="e10">10</xref>) to calculate the loss in the first stage. Where <inline-formula id="inf43">
<mml:math id="m53">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">stg1</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> comprises the pre-trained model&#x2019;s parameters <inline-formula id="inf44">
<mml:math id="m54">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> and MLP model&#x2019;s parameters <inline-formula id="inf45">
<mml:math id="m55">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, with both components jointly optimized during the initial stage; <inline-formula id="inf46">
<mml:math id="m56">
<mml:mi>y</mml:mi>
</mml:math>
</inline-formula> denotes the true labels associated with the input data; <inline-formula id="inf47">
<mml:math id="m57">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula id="inf48">
<mml:math id="m58">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> signify the predicted labels obtained from the model&#x2019;s output corresponding to the original data or component and the shuffled or augmented data component, respectively.</p>
<p>During the initial model training phase, the objective was to optimize parameters across the entire model, entailing adjusting the weights of the complete model aligned with the task&#x2019;s data distribution. However, this process risked compromising some generic features learned by the pre-trained model.</p>
</sec>
<sec id="s3-6">
<title>3.4.2 Second stage: feature-specific fine-tuning with locked pre-trained part</title>
<p>
<disp-formula id="e11">
<mml:math id="m59">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">stg2</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>H</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(11)</label>
</disp-formula>We used Eq. (<xref ref-type="disp-formula" rid="e11">11</xref>) to calculate the loss in the second stage. Where <inline-formula id="inf49">
<mml:math id="m60">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">stg2</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> denotes the MLP model&#x2019;s parameters <inline-formula id="inf50">
<mml:math id="m61">
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>, optimized solely during the second stage, indicating that only the MLP component underwent refinement.</p>
<p>In the subsequent stage, we adopted a different approach by fixing certain components of the pre-trained model, typically the convolutional layers responsible for image feature extraction. Our focus then shifted solely to optimizing the MLP connected downstream. This strategy allowed retraining the generic feature representations using the pre-trained model from extensive data. Subsequently, the MLP underwent fine-tuning to effectively map image features to specific category labels, meeting the requirements of the particular task.</p>
<p>This two-stage method harnessed the advantages of the pre-trained model while tailoring the model&#x2019;s adaptability to task-specific data. This transfer learning strategy is widely employed in computer vision, enhancing model performance even with limited data availability.</p>
</sec>
<sec id="s3-7">
<title>3.5 Implementation details</title>
<p>We utilized PyTorch 2.0.0 and cuDNN 11.7 to implement our model, conducting model training on a single NVIDIA GeForce RTX 4090 GPU with 24&#xa0;GB of memory and a batch size of 76. The Adam optimizer (<xref ref-type="bibr" rid="B23">Kingma and Ba, 2014</xref>) with a base learning rate of 0.0001 and gradient clipping at 1.0 were employed for all the baselines. The model architecture comprises a pre-trained model backbone with the final classification layer removed, a dropout layer with a dropout probability of 0.5 for regularization, and a fully connected linear layer for classification, corresponding to the number of target classes. During forward propagation, input images are processed through the pre-trained model backbone to extract features, which are then flattened into a one-dimensional vector. Dropout regularization is applied, and the resulting features are passed through the fully connected layer to generate the final classification output.</p>
<p>Furthermore, we configured a weight decay of 0.0001 to mitigate overfitting and set the random seed to one for reproducibility. These settings ensure consistent outcomes across different runs. In IDA, adjusting alpha to 0.1 yielded promising model performance. In Stage 1, we conducted training for 10 epochs, fine-tuning the model with a learning rate of 0.0001. In Stage 2, we enhanced model training by extending the duration to 100 epochs and fine-tuned the learning rate to 0.00001.</p>
</sec>
<sec id="s3-8">
<title>3.6 Research questions</title>
<p>This section analyzes the experimental outcomes to demonstrate the efficacy of our proposed IDA and CFT approaches.<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf51">
<mml:math id="m62">
<mml:mo>&#x2022;</mml:mo>
</mml:math>
</inline-formula> <bold>RQ1:</bold> How does the resampling approach impact the clustering effects of pre-trained models when visualizing 2D t-SNE feature embeddings of samples from the &#x2018;head&#x2019; and &#x2018;tail&#x2019; segments of the IP102 dataset?</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf52">
<mml:math id="m63">
<mml:mo>&#x2022;</mml:mo>
</mml:math>
</inline-formula> <bold>RQ2:</bold> How does the L1 loss vary when different subsets (&#x2018;head&#x2019; and &#x2018;tail&#x2019;) of the IP102 dataset are considered, using ResNet18, ResNet50, and ViT as base models with the IDA?</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf53">
<mml:math id="m64">
<mml:mo>&#x2022;</mml:mo>
</mml:math>
</inline-formula> <bold>RQ3:</bold> How does CFT influence the accuracy and F1 score of ResNet-18, ResNet-50, and ViT models on segmented portions (&#x2018;head&#x2019;, &#x2018;mid&#x2019; and &#x2018;tail&#x2019;) of the IP102 dataset, revealing notable enhancements, particularly in the tail segment?</p>
</list-item>
</list>
</p>
<p>We investigated the performance enhancement effects of resampling, IDA, and CFT on pre-trained models separately. In addition, we computed performance metrics including accuracy (Acc), precision (Pre), recall (Rec), F1-score (F1), and geometric mean (GM) for each model. GM evaluated the model&#x2019;s performance in handling class imbalance, with higher GM values indicating better balance and robustness with imbalanced data.</p>
</sec>
</sec>
<sec sec-type="results" id="s4">
<title>4 Results</title>
<sec id="s4-1">
<title>4.1 Experimental results</title>
<p>This study enhanced image classification accuracy on the IP102 dataset by leveraging pre-trained ResNet-18, ResNet-50, and ViT models (<xref ref-type="table" rid="T1">Table 1</xref>). In the subsequent Ablation Study, we explored our model&#x2019;s capability in macro clustering and recognizing tail classes.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>The classification performance of various classifiers using IDA or CFT methods under different evaluation metrics on the IP102 dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">Acc</th>
<th align="center">F1</th>
<th align="center">Pre</th>
<th align="center">Rec</th>
<th align="center">GM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">AlexNet</td>
<td align="center">0.4180</td>
<td align="center">0.3410</td>
<td align="center">-</td>
<td align="center">-</td>
<td align="center">0.2700</td>
</tr>
<tr>
<td align="left">GoogleNet</td>
<td align="center">0.4350</td>
<td align="center">0.3270</td>
<td align="center">-</td>
<td align="center">-</td>
<td align="center">0.2130</td>
</tr>
<tr>
<td align="left">VGGNet</td>
<td align="center">0.4820</td>
<td align="center">0.3870</td>
<td align="center">-</td>
<td align="center">-</td>
<td align="center">0.3090</td>
</tr>
<tr>
<td align="left">ResNet-18</td>
<td align="center">0.5146</td>
<td align="center">0.5367</td>
<td align="center">0.6359</td>
<td align="center">0.5146</td>
<td align="center">0.4266</td>
</tr>
<tr>
<td align="left">ResNet-18(Resample)</td>
<td align="center">0.5675</td>
<td align="center">0.5691</td>
<td align="center">0.5861</td>
<td align="center">0.5675</td>
<td align="center">0.5339</td>
</tr>
<tr>
<td align="left">ResNet-18(Reweight)</td>
<td align="center">0.5270</td>
<td align="center">0.5443</td>
<td align="center">0.6188</td>
<td align="center">0.5270</td>
<td align="center">0.4565</td>
</tr>
<tr>
<td align="left">ResNet-18(Resample &#x2b; IDA)</td>
<td align="center">0.5846</td>
<td align="center">0.5655</td>
<td align="center">0.5611</td>
<td align="center">0.5846</td>
<td align="center">0.5555</td>
</tr>
<tr>
<td align="left">ResNet-18(Resample &#x2b; IDA &#x2b; CFT)</td>
<td align="center">0.5935</td>
<td align="center">0.5876</td>
<td align="center">0.5898</td>
<td align="center">0.5935</td>
<td align="center">0.5659</td>
</tr>
<tr>
<td align="left">ResNet-50</td>
<td align="center">0.5507</td>
<td align="center">0.5614</td>
<td align="center">0.6237</td>
<td align="center">0.5507</td>
<td align="center">0.4927</td>
</tr>
<tr>
<td align="left">ResNet-50 (Resample)</td>
<td align="center">0.5687</td>
<td align="center">0.5638</td>
<td align="center">0.5877</td>
<td align="center">0.5687</td>
<td align="center">0.5330</td>
</tr>
<tr>
<td align="left">ResNet-50 (Reweight)</td>
<td align="center">0.5572</td>
<td align="center">0.5641</td>
<td align="center">0.6090</td>
<td align="center">0.5572</td>
<td align="center">0.4932</td>
</tr>
<tr>
<td align="left">ResNet-50 ((Resample &#x2b; IDA)</td>
<td align="center">0.5906</td>
<td align="center">0.5762</td>
<td align="center">0.5853</td>
<td align="center">0.5906</td>
<td align="center">0.5534</td>
</tr>
<tr>
<td align="left">ResNet-50(Resample &#x2b; IDA &#x2b; CFT)</td>
<td align="center">0.6200</td>
<td align="center">
<bold>0.6323</bold>
</td>
<td align="center">
<bold>0.6542</bold>
</td>
<td align="center">0.6200</td>
<td align="center">0.5883</td>
</tr>
<tr>
<td align="left">ViT</td>
<td align="center">0.5853</td>
<td align="center">0.5938</td>
<td align="center">0.6492</td>
<td align="center">0.5853</td>
<td align="center">0.5091</td>
</tr>
<tr>
<td align="left">ViT (Resample)</td>
<td align="center">0.6039</td>
<td align="center">0.5665</td>
<td align="center">0.5583</td>
<td align="center">0.6039</td>
<td align="center">0.5710</td>
</tr>
<tr>
<td align="left">ViT (Reweight)</td>
<td align="center">0.5924</td>
<td align="center">0.5985</td>
<td align="center">0.6410</td>
<td align="center">0.5924</td>
<td align="center">0.5356</td>
</tr>
<tr>
<td align="left">ViT ((Resample &#x2b; IDA)</td>
<td align="center">0.6304</td>
<td align="center">0.5886</td>
<td align="center">0.5719</td>
<td align="center">0.6304</td>
<td align="center">0.6005</td>
</tr>
<tr>
<td align="left">ViT (Resample &#x2b; IDA &#x2b; CFT)</td>
<td align="center">
<bold>0.6426</bold>
</td>
<td align="center">0.6179</td>
<td align="left">0.6037</td>
<td align="center">
<bold>0.6426</bold>
</td>
<td align="center">
<bold>0.6174</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values means the performance of the approach.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Then, we employed the IDA &#x2b; CFT approach using pre-trained ResNet-18, ResNet-50, and ViT models on the CIFAR-10-LT and CIFAR-100-LT datasets, variants of the CIFAR-10 and CIFAR-100 datasets, respectively, with long-tailed class distributions. Our experimental results indicated that our proposed approach enhanced the models&#x2019; classification performance. We achieved optimal performance with ViT pre-trained models as the base and employing our proposed method (<xref ref-type="table" rid="T2">Table 2</xref>). Our proposed method architecture outperforms pre-trained models on CIFAR-10-LT and CIFAR-100-LT. Moreover, our evaluation of the CIFAR-10-LT and CIFAR-100-LT datasets has demonstrated state-of-the-art performance.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Top-1 accuracy (%) on CIFAR-10-LT and CIFAR-100-LT with different imbalance factors [100, 50, 10].</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Method</th>
<th colspan="3" align="left">CIFAR-10-LT</th>
<th colspan="3" align="left">CIFAR-100-LT</th>
</tr>
<tr>
<th align="left"/>
<th align="left">IF &#x3d; 100</th>
<th align="left">50</th>
<th align="left">10</th>
<th align="left">IF &#x3d; 100</th>
<th align="left">50</th>
<th align="left">10</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">CB-Focal (<xref ref-type="bibr" rid="B9">Cui et al., 2019</xref>)</td>
<td align="left">74.60</td>
<td align="left">79.30</td>
<td align="left">87.10</td>
<td align="left">39.60</td>
<td align="left">45.20</td>
<td align="left">58.00</td>
</tr>
<tr>
<td align="left">BBN (<xref ref-type="bibr" rid="B66">Zhou et al., 2020b</xref>)</td>
<td align="left">79.82</td>
<td align="left">82.18</td>
<td align="left">88.32</td>
<td align="left">42.56</td>
<td align="left">47.02</td>
<td align="left">59.12</td>
</tr>
<tr>
<td align="left">LogitAjust <xref ref-type="bibr" rid="B34">Menon et al. (2020)</xref>
</td>
<td align="left">80.92</td>
<td align="left">-</td>
<td align="left">-</td>
<td align="left">42.01</td>
<td align="left">47.03</td>
<td align="left">57.74</td>
</tr>
<tr>
<td align="left">RISDA (<xref ref-type="bibr" rid="B4">Chen et al., 2022</xref>)</td>
<td align="left">79.89</td>
<td align="left">79.89</td>
<td align="left">79.89</td>
<td align="left">50.16</td>
<td align="left">53.84</td>
<td align="left">62.38</td>
</tr>
<tr>
<td align="left">MiSLAS (<xref ref-type="bibr" rid="B65">Zhong et al., 2021</xref>)</td>
<td align="left">82.10</td>
<td align="left">85.70</td>
<td align="left">90.00</td>
<td align="left">47.00</td>
<td align="left">52.30</td>
<td align="left">63.20</td>
</tr>
<tr>
<td align="left">GLMC (<xref ref-type="bibr" rid="B17">Du et al., 2023</xref>)</td>
<td align="left">94.18</td>
<td align="left">95.13</td>
<td align="left">95.70</td>
<td align="left">57.11</td>
<td align="left">62.32</td>
<td align="left">72.33</td>
</tr>
<tr>
<td align="left">ResNet-18</td>
<td align="left">61.06</td>
<td align="left">69.21</td>
<td align="left">75.99</td>
<td align="left">35.63</td>
<td align="left">38.33</td>
<td align="left">49.30</td>
</tr>
<tr>
<td align="left">ResNet-18 (IDA)</td>
<td align="left">64.82</td>
<td align="left">71.39</td>
<td align="left">77.90</td>
<td align="left">37.68</td>
<td align="left">41.04</td>
<td align="left">50.90</td>
</tr>
<tr>
<td align="left">ResNet-18 (IDA &#x2b; CFT)</td>
<td align="left">67.05</td>
<td align="left">73.09</td>
<td align="left">80.85</td>
<td align="left">38.51</td>
<td align="left">41.54</td>
<td align="left">51.25</td>
</tr>
<tr>
<td align="left">ResNet-50</td>
<td align="left">66.50</td>
<td align="left">72.78</td>
<td align="left">80.65</td>
<td align="left">38.50</td>
<td align="left">41.84</td>
<td align="left">52.00</td>
</tr>
<tr>
<td align="left">ResNet-50 (IDA)</td>
<td align="left">66.54</td>
<td align="left">74.24</td>
<td align="left">82.64</td>
<td align="left">39.48</td>
<td align="left">43.30</td>
<td align="left">53.76</td>
</tr>
<tr>
<td align="left">ResNet-50 (IDA &#x2b; CFT)</td>
<td align="left">69.50</td>
<td align="left">74.92</td>
<td align="left">84.12</td>
<td align="left">40.73</td>
<td align="left">44.62</td>
<td align="left">56.22</td>
</tr>
<tr>
<td align="left">ViT</td>
<td align="left">95.17</td>
<td align="left">96.14</td>
<td align="left">97.47</td>
<td align="left">73.04</td>
<td align="left">78.40</td>
<td align="left">85.07</td>
</tr>
<tr>
<td align="left">ViT (IDA)</td>
<td align="left">96.12</td>
<td align="left">97.17</td>
<td align="left">98.01</td>
<td align="left">77.00</td>
<td align="left">81.81</td>
<td align="left">87.05</td>
</tr>
<tr>
<td align="left">ViT (IDA &#x2b; CFT)</td>
<td align="left">
<bold>96.75</bold>
</td>
<td align="left">
<bold>97.57</bold>
</td>
<td align="left">
<bold>98.21</bold>
</td>
<td align="left">
<bold>80.00</bold>
</td>
<td align="left">
<bold>84.13</bold>
</td>
<td align="left">
<bold>88.62</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The bold values means the performance of the approach.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Based on pre-trained ResNet-18, ResNet-50, and ViT models, we implemented the resample and reweight methods, training and evaluating the models. Our findings indicate improvements in accuracy, F1-score, precision, recall, and G-Mean across all three models. The resampling approach exhibited a superior enhancement in performance due to its ability to balance sample distribution among classes. This mechanism mitigated the impact of class imbalance during model training and facilitated more effective learning and classification of samples from each class.</p>
<p>Overall, implementing the resampling and reweighting approaches boosted the performance of the three models in image classification tasks. Based on the pre-trained ResNet-18 model, we combined the resampling method with IDA in our experiments, yielding significant improvements in model performance. Incorporating IDA and the resampling method notably improved accuracy, F1 score, recall, and G-Mean, albeit with a slight decrease in precision. Furthermore, integrating CFT with the aforementioned approach further enhanced performance across various metrics.</p>
<p>Similarly, utilizing the pre-trained ResNet-50 model, we conducted experiments employing the resample &#x2b; IDA &#x2b; CFT approach, significantly enhancing model performance. Compared to the ResNet-50 and ResNet-50 (resample) models, we achieved approximately a 7% increase in accuracy and a 6% increase in precision, respectively. Furthermore, integrating the CFT method into the aforementioned approach yielded further improvements across various metrics. Incorporating this method yielded an additional 3% increase in precision. Moreover, the CFT method rectified the trade-off in precision made by the IDA method, leading to a 7% increase in precision compared to not using the CFT method. Utilizing the pre-trained ViT model, we employed the resample &#x2b; IDA &#x2b; CFT method, similar to the approach with ResNet-50. An accuracy of 64.26% and a G-mean of 61.74% was observed. Unlike ResNet-50, where all metrics improved, our approach with ViT yielded similar results as ResNet-18, enhancing accuracy, F1-score, recall, and G-mean while compromising precision.</p>
</sec>
<sec id="s4-2">
<title>4.2 Effectiveness of resampling</title>
<p>We partitioned the IP102 dataset into two segments: &#x2018;head&#x2019; and &#x2018;tail&#x2019;. Within each segment, we randomly selected five categories and conducted random sampling to obtain five samples per category, resulting in 50 samples. In <xref ref-type="fig" rid="F4">Figures 4</xref>&#x2013;<xref ref-type="fig" rid="F6">6</xref>, we visualized the distribution of these samples on the IP102 dataset using 2D t-SNE feature embeddings.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Comparison of t-SNE visualizations with and without data resampling during fine-tuning of ResNet-18.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g004.tif"/>
</fig>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Comparison of t-SNE visualizations with and without data resampling during fine-tuning of ResNet-50.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g005.tif"/>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Comparison of t-SNE visualizations with and without data resampling during fine-tuning of ViT.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g006.tif"/>
</fig>
<p>By comparing the results of utilizing the resampling method with not using it on pre-trained models, we discerned differences by plotting t-SNE graphs. Our findings exhibit more pronounced clustering effects in models using the resampling approach, with particularly significant improvements observed in the ViT model.</p>
</sec>
<sec id="s4-3">
<title>4.3 Effect on IDA</title>
<p>After using the previously mentioned approach of segmenting the IP102 and generating 50 samples, we employed the IDA method to encompass three scenarios: head combined with head, head combined with tail, and tail combined with tail. We visualized the L1 loss in these scenarios as bar graphs (<xref ref-type="fig" rid="F7">Figure 7</xref>).</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Fine-tuned ResNet-18 model shows that post-IDA, L1 loss performs best with head and tail combination, and worst with head and head combination. Fine-tuned ResNet-50 model shows that post-IDA, L1 loss performs best with head and tail combination, and worst with tail and tail combination. Fine-tuned ViT model exhibits that post-IDA, L1 loss performs best with head and tail combination, and worst with tail and tail combination.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g007.tif"/>
</fig>
<p>The head and tail data combination yielded the best performance in terms of L1 loss. Across ResNet18 and ResNet50 models, the tail and tail data combination consistently outperformed the head and head data combination in terms of L1 loss. Overall, the ResNet-50 model exhibited the lowest average L1 loss among the three combinations.</p>
</sec>
<sec id="s4-4">
<title>4.4 Effect on CFA</title>
<p>We partitioned the IP102 dataset into three segments: &#x2018;head&#x2019;, &#x2018;mid&#x2019; and &#x2018;tail&#x2019;. Within each segment, we randomly selected five categories and conducted random sampling to obtain five samples per category, resulting in 75 samples. Utilizing the resampling and IDA as the base model, we evaluated the impact of CFT on the model&#x2019;s accuracy and F1 score. <xref ref-type="fig" rid="F8">Figures 8</xref>&#x2013;<xref ref-type="fig" rid="F10">10</xref> illustrate that the CFT positively affected the accuracy and F1-score of the head segment in both the ResNet-50 and ViT models. Moreover, the CFT method significantly improved the tail segment for both models. Additionally, the ViT model demonstrated enhancements across all three categories, with the most noticeable performance boost observed in the mid portion of the data.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Fine-tuned ResNet-18 model&#x2019;s accuracy and F1-score with and without utilizing CFT.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g008.tif"/>
</fig>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Fine-tuned ResNet-50 model&#x2019;s accuracy and F1-score with and without using CFT.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g009.tif"/>
</fig>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Fine-tuned ViT model&#x2019;s accuracy and F1-score with and without using CFT.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g010.tif"/>
</fig>
</sec>
</sec>
<sec id="s5">
<title>5 Case study</title>
<p>We segmented the dataset into three segments: &#x2018;head&#x2019;, &#x2018;mid&#x2019; and &#x2018;tail&#x2019;. Within each part, we sampled five categories and selected five images per category, totaling 75 images. To determine the specific distribution probabilities of these images across categories, we employed the ViT (Resample &#x2b; IDA &#x2b; CFT) methodology. The specific sampling probability distribution is shown in <xref ref-type="fig" rid="F11">Figure 11</xref>.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>The specific sampling probability distribution of the head, mid, and tail parts.</p>
</caption>
<graphic xlink:href="fenvs-12-1391770-g011.tif"/>
</fig>
<p>In the three scenarios, a relatively simple background or a significant contrast between insect colors and the background facilitated easier target object detection, leading to more accurate classification and recognition.</p>
<p>However, when the color of the target object closely resembled that of the background or when blurriness was present, the model&#x2019;s recognition capability was compromised. The similarity in colors can blurred the boundaries between the target object and the background. In such instances, the model exhibited diminished recognition capability or even failed to identify the target accurately.</p>
</sec>
<sec sec-type="discussion" id="s6">
<title>6 Discussion</title>
<p>Previous studies on pest image recognition often suffer from the long-tail distribution problem, leading to insufficient samples for rare pest categories and subsequently affecting the overall performance of the models. Many researchers have proposed various methods to address this issue, but these methods largely focus on variations of neural network architectures (<xref ref-type="bibr" rid="B55">Ung et al., 2021</xref>; <xref ref-type="bibr" rid="B6">Coulibaly et al., 2022b</xref>; <xref ref-type="bibr" rid="B36">Nanni et al., 2022</xref>), introducing new architectures to tackle the problem. These approaches often have specific prerequisites, such as recognizing certain pests (<xref ref-type="bibr" rid="B49">Sethy et al., 2020</xref>; <xref ref-type="bibr" rid="B26">Li S. et al., 2022</xref>) or targeting pest recognition in specific scenarios (<xref ref-type="bibr" rid="B47">Sankaran et al., 2015</xref>). The pest species identified in these studies are typically on a smaller scale. However, our proposed method is based on the IP102 large-scale dataset for pest recognition, which presents a natural long-tail distribution, making it more representative of real-world conditions.</p>
<p>IDA integrates resampling techniques to ensure equitable representation for underrepresented categories, thereby alleviating bias stemming from imbalanced sample distributions. Additionally, it incorporates self-attention mechanisms to discern intricate relationships among samples, thereby enhancing classification performance across all categories. On the other hand, CFT focuses on optimizing a limited subset of model parameters post-feature extraction, particularly enriching the performance of the MLP layer for more precise feature classification and generalization.</p>
<p>The overall model performance improvement can be attributed to IDA&#x2019;s ability to rebalance the dataset, providing fairer representation for tail-end categories without compromising the performance of common samples. Moreover, the selective optimization of model parameters by CFT post-feature extraction aids in preventing overfitting and enhancing the model&#x2019;s ability to generalize features to actual labels, particularly benefiting minority classes.</p>
<p>Using ViT pre-trained models as baselines on the IP102 dataset, we observed a 5.73% improvement in accuracy, a 2.41% improvement in F1 score, and a 10.83% improvement in GM. In CIFAR-10-LT (IF &#x3d; 50) and CIFAR-100-LT (IF &#x3d; 50), our models achieved Top-1 accuracies of 97.57% and 84.13%, respectively. Compared to the latest research results, our models demonstrate state-of-the-art performance.</p>
</sec>
<sec sec-type="conclusion" id="s7">
<title>7 Conclusion</title>
<p>This study introduces two cutting-edge techniques, IDA and CFT, to tackle the challenges posed by long-tail image classification tasks in the realm of pest distribution research. Traditionally, such studies have been hindered by the long-tail distribution issue, resulting in insufficient samples for rare pest categories and thereby impacting overall model performance. While previous approaches have mainly focused on neural network architecture variations to address this issue, our proposed methods are rooted in data augmentation and feature mapping.</p>
<p>The combined use of IDA and CFT yielded superior accuracy and F1 score, particularly for the tail-end classes. Our experimental results on the IP102 dataset revealed significant improvements in long-tail image classification tasks by integrating IDA and CFT. In particular, leveraging ResNet-18, ResNet-50, and ViT as baseline pre-trained models, implementing IDA and CFT approaches increased accuracy by 7.89%, 6.93%, and 5.73%, respectively. Our experiments indicated that the IDA and CFT methods bolstered overall accuracy without compromising the accuracy of classes with abundant samples by elevating accuracy across the dataset in the tail and head sections. These methods have the potential to substantially enhance the robustness and accuracy of models in real-world applications.</p>
<p>We envision that integrating IDA and CFT will promote advancements in image classification and provide valuable insights for addressing complex and imbalanced data challenges across diverse domains. Our research lays the groundwork for developing more accurate and reliable recognition systems, particularly involving long-tail distributions in image data.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s9">
<title>Author contributions</title>
<p>SC: Conceptualization, Formal Analysis, Methodology, Software, Validation, Visualization, Writing&#x2013;original draft, Data curation, Investigation, Writing&#x2013;review and editing. QG: Conceptualization, Formal Analysis, Funding acquisition, Methodology, Project administration, Resources, Supervision, Validation, Writing&#x2013;review and editing. YH: Conceptualization, Data curation, Formal Analysis, Funding acquisition, Investigation, Project administration, Resources, Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This research was funded by National Natural Science Foundation of China (Grant no. 32101611), the Youth Project of Basic Research Program of Yunnan Province (Grant no. 202101AU070096), the Major Project of Science and Technology of Yunnan Province (Grant nos. 202202AE090021 and 202302AE090020), and the Open Research Program of State Key Laboratory for Conservation and Utilization of Bio-Resource in Yunnan (Grant no. GZKF2021009).</p>
</sec>
<ack>
<p>Thanks to the editorial team and the reviewers for their valuable feedback and assistance in improving our article.</p>
</ack>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Antoniou</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Storkey</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Edwards</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Data augmentation generative adversarial networks</article-title>. <source>Corr. abs/1711.04340</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1711.04340</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bataa</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>An investigation of transfer learning-based sentiment analysis in Japanese</article-title>,&#x201d; in <conf-name>Proceedings of the 57th Conference of the Association for Computational Linguistics, ACL 2019</conf-name>, <conf-loc>Florence, Italy</conf-loc>, <conf-date>July 28- August 2, 2019</conf-date>. Editors <person-group person-group-type="editor">
<name>
<surname>Korhonen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Traum</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>M&#xe0;rquez</surname>
<given-names>L.</given-names>
</name>
</person-group> (<publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>4652</fpage>&#x2013;<lpage>4657</lpage>. <comment>Volume 1: Long Papers</comment>. <pub-id pub-id-type="doi">10.18653/v1/p19-1458</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chaitanya</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Erdil</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Karani</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Konukoglu</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Contrastive learning of global and local features for medical image segmentation with limited annotations</article-title>,&#x201d; in <conf-name>Advances in Neural Information Processing Systems 33: Annual Conference on Neural Information Processing Systems 2020, NeurIPS 2020</conf-name>, <conf-date>December 6-12, 2020</conf-date>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Imagine by reasoning: a reasoning-based implicit semantic data augmentation for long-tailed classification</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>36</volume>, <fpage>356</fpage>&#x2013;<lpage>364</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v36i1.19912</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Coulibaly</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kamsu-Foguem</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kamissoko</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Traore</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Explainable deep convolutional neural networks for insect pest recognition</article-title>. <source>J. Clean. Prod.</source> <volume>371</volume>, <fpage>133638</fpage>. <pub-id pub-id-type="doi">10.1016/j.jclepro.2022.133638</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Coulibaly</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kamsu-Foguem</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kamissoko</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Traore</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Explainable deep convolutional neural networks for insect pest recognition</article-title>. <source>J. Clean. Prod.</source> <volume>371</volume>, <fpage>133638</fpage>. <pub-id pub-id-type="doi">10.1016/j.jclepro.2022.133638</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cubuk</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>Zoph</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Man&#xe9;</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Vasudevan</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q. V.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Autoaugment: learning augmentation strategies from data</article-title>,&#x201d; in <conf-name>IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2019</conf-name>, <conf-loc>Long Beach, CA, USA</conf-loc>, <conf-date>June 16-20, 2019</conf-date> (<publisher-name>Computer Vision Foundation/IEEE</publisher-name>), <fpage>113</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00020</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cubuk</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>Zoph</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Shlens</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q. V.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Randaugment: practical automated data augmentation with a reduced search space</article-title>,&#x201d; in <conf-name>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition, CVPR Workshops 2020</conf-name>, <conf-loc>Seattle, WA, USA</conf-loc>, <conf-date>June 14-19, 2020</conf-date> (<publisher-name>Computer Vision Foundation/IEEE</publisher-name>), <fpage>3008</fpage>&#x2013;<lpage>3017</lpage>. <pub-id pub-id-type="doi">10.1109/CVPRW50498.2020.00359</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cui</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>T.-Y.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Belongie</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Class-balanced loss based on effective number of samples</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>, <fpage>9268</fpage>&#x2013;<lpage>9277</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>UP-DETR: unsupervised pre-training for object detection with transformers</article-title>,&#x201d; in <conf-name>IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2021, virtual</conf-name>, <conf-date>June 19-25, 2021</conf-date> (<publisher-name>Computer Vision Foundation/IEEE</publisher-name>), <fpage>1601</fpage>&#x2013;<lpage>1610</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR46437.2021.00165</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dalal</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Triggs</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Histograms of oriented gradients for human detection</article-title>,&#x201d; in <conf-name>2005 IEEE Computer Society Conference on Computer Vision and Pattern Recognition (CVPR 2005)</conf-name>, <conf-loc>20-26 June 2005</conf-loc>, <conf-date>San Diego, CA, USA</conf-date> (<publisher-name>IEEE Computer Society</publisher-name>), <fpage>886</fpage>&#x2013;<lpage>893</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2005.177</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Boer</surname>
<given-names>P.-T.</given-names>
</name>
<name>
<surname>Kroese</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Mannor</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rubinstein</surname>
<given-names>R. Y.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>A tutorial on the cross-entropy method</article-title>. <source>Ann. operations Res.</source> <volume>134</volume>, <fpage>19</fpage>&#x2013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1007/s10479-005-5724-z</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Devlin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Toutanova</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>BERT: pre-training of deep bidirectional transformers for language understanding</article-title>,&#x201d; in <conf-name>Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, NAACL-HLT 2019</conf-name>, <conf-loc>Minneapolis, MN, USA</conf-loc>, <conf-date>June 2-7, 2019</conf-date>. Editors <person-group person-group-type="editor">
<name>
<surname>Burstein</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Doran</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Solorio</surname>
<given-names>T.</given-names>
</name>
</person-group> (<publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>4171</fpage>&#x2013;<lpage>4186</lpage>. <comment>Volume 1 (Long and Short Papers)</comment>. <pub-id pub-id-type="doi">10.18653/v1/n19-1423</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devries</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Taylor</surname>
<given-names>G. W.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Improved regularization of convolutional neural networks with <italic>cutout</italic>
</article-title>. <source>Corr. abs/1708</source>, <fpage>04552</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1708.04552</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dewi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Christanto</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Automated identification of insect pests: a deep transfer learning approach using resnet</article-title>. <source>Acadlore Trans. Mach. Learn</source> <volume>2</volume>, <fpage>194</fpage>&#x2013;<lpage>203</lpage>. <pub-id pub-id-type="doi">10.56578/ataiml020402</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dosovitskiy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Beyer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kolesnikov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Weissenborn</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Unterthiner</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>An image is worth 16x16 words: transformers for image recognition at scale</article-title>. <source>Corr. abs/2010</source>, <fpage>11929</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Du</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Nan</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Global and local mixture consistency cumulative learning for long-tailed visual recognitions</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>, <fpage>15814</fpage>&#x2013;<lpage>15823</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dyrmann</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Karstoft</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Midtiby</surname>
<given-names>H. S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Plant species classification using deep convolutional neural network</article-title>. <source>Biosyst. Eng.</source> <volume>151</volume>, <fpage>72</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1016/j.biosystemseng.2016.08.024</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goodfellow</surname>
<given-names>I. J.</given-names>
</name>
<name>
<surname>Pouget-Abadie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Mirza</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Warde-Farley</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ozair</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Generative adversarial networks</article-title>. <source>Corr. abs/1406.2661</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1406.2661</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <source>Deep residual learning for image recognition</source>, <fpage>03385</fpage>. <comment>CoRR abs/1512</comment>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>J.-F.</given-names>
</name>
<name>
<surname>Apan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Detection of sclerotinia rot disease on celery using hyperspectral data and partial least squares regression</article-title>. <source>J. Spatial Sci.</source> <volume>51</volume>, <fpage>129</fpage>&#x2013;<lpage>142</lpage>. <pub-id pub-id-type="doi">10.1080/14498596.2006.9635087</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kasinathan</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Reddy</surname>
<given-names>U. S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Crop pest classification based on deep convolutional neural network and transfer learning</article-title>. <source>Comput. Electron. Agric.</source> <volume>164</volume>, <fpage>104906</fpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2019.104906</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <source>Adam: a method for stochastic optimization</source>. <comment>arXiv preprint arXiv:1412.6980</comment>.</citation>
</ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Krizhevsky</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G. E.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Imagenet classification with deep convolutional neural networks</article-title>,&#x201d; in <conf-name>Advances in Neural Information Processing Systems 25: 26th Annual Conference on Neural Information Processing Systems 2012</conf-name>, <conf-loc>Proceedings of a meeting held December 3-6, 2012</conf-loc>, <conf-date>Lake Tahoe, Nevada, United States</conf-date>. Editors <person-group person-group-type="editor">
<name>
<surname>Bartlett</surname>
<given-names>P. L.</given-names>
</name>
<name>
<surname>Pereira</surname>
<given-names>F. C. N.</given-names>
</name>
<name>
<surname>Burges</surname>
<given-names>C. J. C.</given-names>
</name>
<name>
<surname>Bottou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Weinberger</surname>
<given-names>K. Q.</given-names>
</name>
</person-group>, <fpage>1106</fpage>&#x2013;<lpage>1114</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kusrini</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Suputa</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Setyanto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Agastya</surname>
<given-names>I. M. A.</given-names>
</name>
<name>
<surname>Priantoro</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chandramouli</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>Data augmentation for automated pest classification in mango farms</article-title>,&#x201d;, <fpage>105842</fpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105842</pub-id>
<source>Comput. Electron. Agric.</source>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>A self-attention feature fusion model for rice pest detection</article-title>. <source>IEEE Access</source> <volume>10</volume>, <fpage>84063</fpage>&#x2013;<lpage>84077</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3194925</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Recommending advanced deep learning models for efficient insect pest detection</article-title>. <source>Agriculture</source> <volume>12</volume>, <fpage>1065</fpage>. <pub-id pub-id-type="doi">10.3390/agriculture12071065</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Dice loss for data-imbalanced NLP tasks</article-title>,&#x201d; in <conf-name>Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, ACL 2020, Online</conf-name>, <conf-date>July 5-10, 2020</conf-date>. Editors <person-group person-group-type="editor">
<name>
<surname>Jurafsky</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chai</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schluter</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Tetreault</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>465</fpage>&#x2013;<lpage>476</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2020.acl-main.45</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname>
<given-names>R. B.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Focal loss for dense object detection</article-title>,&#x201d; in <conf-name>IEEE International Conference on Computer Vision, ICCV 2017</conf-name>, <conf-loc>Venice, Italy</conf-loc>, <conf-date>October 22-29, 2017</conf-date> (<publisher-name>IEEE Computer Society</publisher-name>), <fpage>2999</fpage>&#x2013;<lpage>3007</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2017.324</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Dff-resnet: an insect pest recognition model based on residual networks</article-title>. <source>Big Data Min. Anal.</source> <volume>3</volume>, <fpage>300</fpage>&#x2013;<lpage>310</lpage>. <pub-id pub-id-type="doi">10.26599/BDMA.2020.9020021</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Forest pest identification based on a new dataset and convolutional neural network model with enhancement strategy</article-title>. <source>Comput. Electron. Agric.</source> <volume>192</volume>, <fpage>106625</fpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2021.106625</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cahyawijaya</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Crossner: evaluating cross-domain named entity recognition</article-title>,&#x201d; in <conf-name>Thirty-Fifth AAAI Conference on Artificial Intelligence, AAAI 2021, Thirty-Third Conference on Innovative Applications of Artificial Intelligence, IAAI 2021, The Eleventh Symposium on Educational Advances in Artificial Intelligence, EAAI 2021, Virtual Event</conf-name>, <conf-date>February 2-9, 2021</conf-date> (<publisher-loc>Palo Alto, CA, United States</publisher-loc>: <publisher-name>AAAI Press</publisher-name>), <fpage>13452</fpage>&#x2013;<lpage>13460</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v35i15.17587</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lowe</surname>
<given-names>D. G.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Distinctive image features from scale-invariant keypoints</article-title>. <source>Int. J. Comput. Vis.</source> <volume>60</volume>, <fpage>91</fpage>&#x2013;<lpage>110</lpage>. <pub-id pub-id-type="doi">10.1023/B:VISI.0000029664.99615.94</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Menon</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Jayasumana</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rawat</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Veit</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Long-tail learning via logit adjustment</source>.</citation>
</ref>
<ref id="B35">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Mun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>D. K.</given-names>
</name>
<name>
<surname>Ko</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Generative adversarial network based acoustic scene training set augmentation and selection using SVM hyper-plane</article-title>,&#x201d; in <conf-name>Proceedings of the Workshop on Detection and Classification of Acoustic Scenes and Events, DCASE 2017</conf-name>, <conf-loc>Munich, Germany</conf-loc>, <conf-date>November 16-17, 2017</conf-date>. Editors <person-group person-group-type="editor">
<name>
<surname>Virtanen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Mesaros</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Heittola</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Diment</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vincent</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Benetos</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> <fpage>93</fpage>&#x2013;<lpage>102</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nanni</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Manf&#xe8;</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maguolo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lumini</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brahnam</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>High performing ensemble of convolutional neural networks for insect pest image detection</article-title>. <source>Ecol. Inf.</source> <volume>67</volume>, <fpage>101515</fpage>. <pub-id pub-id-type="doi">10.1016/j.ecoinf.2021.101515</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oerke</surname>
<given-names>E.-C.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Crop losses to pests</article-title>. <source>J. Agric. Sci.</source> <volume>144</volume>, <fpage>31</fpage>&#x2013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1017/s0021859605005708</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Patel</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bhatt</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Improved accuracy of pest detection using augmentation approach with faster r-cnn</article-title>. <source>IOP Publ.</source> <volume>1042</volume>, <fpage>012020</fpage>. <pub-id pub-id-type="doi">10.1088/1757-899x/1042/1/012020</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Perez</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>The effectiveness of data augmentation in image classification using deep learning</article-title>. <source>Corr. abs/1712</source>, <fpage>04621</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1712.04621</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Poth</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pfeiffer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>R&#xfc;ckl&#xe9;</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gurevych</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>What to pre-train on? efficient intermediate task selection</article-title>,&#x201d; in <conf-name>Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing, EMNLP 2021, Virtual Event/Online and Punta Cana</conf-name>, <conf-loc>Dominican Republic</conf-loc>, <conf-date>7-11 November, 2021</conf-date>. Editors <person-group person-group-type="editor">
<name>
<surname>Moens</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Specia</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yih</surname>
<given-names>S. W.</given-names>
</name>
</person-group> (<publisher-name>Association for Computational Linguistics</publisher-name>), <fpage>10585</fpage>&#x2013;<lpage>10605</lpage>. <pub-id pub-id-type="doi">10.18653/v1/2021.emnlp-main.827</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qian</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An effective pest detection method with automatic data augmentation strategy in the agricultural field</article-title>. <source>Signal, Image Video Process.</source> <volume>17</volume>, <fpage>563</fpage>&#x2013;<lpage>571</lpage>. <pub-id pub-id-type="doi">10.1007/s11760-022-02261-9</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Narasimhan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Salimans</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Sutskever</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Improving language understanding by generative pre-training</article-title>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rani</surname>
<given-names>R. U.</given-names>
</name>
<name>
<surname>Amsini</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Pest identification in leaf images using svm classifier</article-title>. <source>Int. J. Comput. Intell. Inf.</source> <volume>6</volume>, <fpage>248</fpage>&#x2013;<lpage>260</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Feature reuse residual networks for insect pest recognition</article-title>. <source>IEEE Access</source> <volume>7</volume>, <fpage>122758</fpage>&#x2013;<lpage>122768</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2938194</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Salih</surname>
<given-names>T. A.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Ahmed</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Deep learning convolution neural network to detect and classify tomato plant leaf diseases</source>. <publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B46">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Samanta</surname>
<given-names>R. K.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2012</year>). <source>Tea insect pests classification based on artificial neural networks</source>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sankaran</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Khot</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>Espinoza</surname>
<given-names>C. Z.</given-names>
</name>
<name>
<surname>Jarolmasjed</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sathuvalli</surname>
<given-names>V. R.</given-names>
</name>
<name>
<surname>Vandemark</surname>
<given-names>G. J.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Low-altitude, high-resolution aerial imaging systems for row and field crop phenotyping: a review</article-title>. <source>Eur. J. Agron.</source> <volume>70</volume>, <fpage>112</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1016/j.eja.2015.07.004</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sato</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Nishimura</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yokoi</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>APAC: augmented pattern classification with neural networks</article-title>. <source>Corr. abs/1505</source>, <fpage>03229</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1505.03229</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sethy</surname>
<given-names>P. K.</given-names>
</name>
<name>
<surname>Barpanda</surname>
<given-names>N. K.</given-names>
</name>
<name>
<surname>Rath</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Behera</surname>
<given-names>S. K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep feature based rice leaf disease identification using support vector machine</article-title>. <source>Comput. Electron. Agric.</source> <volume>175</volume>, <fpage>105527</fpage>. <pub-id pub-id-type="doi">10.1016/j.compag.2020.105527</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shafri</surname>
<given-names>H. Z.</given-names>
</name>
<name>
<surname>Hamdan</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Hyperspectral imagery for mapping disease infection in oil palm plantationusing vegetation indices and red edge techniques</article-title>. <source>Am. J. Appl. Sci.</source> <volume>6</volume>, <fpage>1031</fpage>&#x2013;<lpage>1035</lpage>. <pub-id pub-id-type="doi">10.3844/ajassp.2009.1031.1035</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Simard</surname>
<given-names>P. Y.</given-names>
</name>
<name>
<surname>Steinkraus</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Platt</surname>
<given-names>J. C.</given-names>
</name>
</person-group> (<year>2003</year>). &#x201c;<article-title>Best practices for convolutional neural networks applied to visual document analysis</article-title>,&#x201d; in <conf-name>7th International Conference on Document Analysis and Recognition (ICDAR 2003), 2-Volume Set</conf-name>, <conf-loc>Edinburgh, Scotland, UK</conf-loc>, <conf-date>3-6 August 2003</conf-date> (<publisher-name>IEEE Computer Society</publisher-name>), <fpage>958</fpage>&#x2013;<lpage>962</lpage>. <pub-id pub-id-type="doi">10.1109/ICDAR.2003.1227801</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zisserman</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Very deep convolutional networks for large-scale image recognition</article-title>,&#x201d; in <conf-name>3rd International Conference on Learning Representations, ICLR 2015</conf-name>, <conf-loc>San Diego, CA, USA</conf-loc>, <conf-date>May 7-9, 2015</conf-date>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Spinelli</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Noferini</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Costa</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Near infrared spectroscopy (nirs): perspective of fire blight detection in asymptomatic plant material</article-title>. <source>X Int. Workshop Fire Blight</source> <volume>704</volume>, <fpage>87</fpage>&#x2013;<lpage>90</lpage>. <pub-id pub-id-type="doi">10.17660/actahortic.2006.704.9</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sermanet</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Reed</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Anguelov</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). &#x201c;<article-title>Going deeper with convolutions</article-title>,&#x201d; in <conf-name>IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2015</conf-name>, <conf-loc>Boston, MA, USA</conf-loc>, <conf-date>June 7-12, 2015</conf-date> (<publisher-name>IEEE Computer Society</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2015.7298594</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ung</surname>
<given-names>H. T.</given-names>
</name>
<name>
<surname>Ung</surname>
<given-names>H. Q.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>B. T.</given-names>
</name>
</person-group> (<year>2021</year>). <source>An efficient insect pest classification using multiple convolutional neural network based models</source>. <comment>arXiv preprint arXiv:2107.12189</comment>.</citation>
</ref>
<ref id="B56">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). &#x201c;<article-title>Attention is all you need</article-title>,&#x201d; in <conf-name>Advances in Neural Information Processing Systems 30: Annual Conference on Neural Information Processing Systems 2017</conf-name>, <conf-loc>December 4-9, 2017</conf-loc>, <conf-date>Long Beach, CA, USA</conf-date>. Editors <person-group person-group-type="editor">
<name>
<surname>Guyon</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>von Luxburg</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wallach</surname>
<given-names>H. M.</given-names>
</name>
<name>
<surname>Fergus</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Vishwanathan</surname>
<given-names>S. V. N.</given-names>
</name>
<etal/>
</person-group> <fpage>5998</fpage>&#x2013;<lpage>6008</lpage>.</citation>
</ref>
<ref id="B57">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wan</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zeiler</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fergus</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Regularization of neural networks using dropconnect</article-title>,&#x201d; in <conf-name>Proceedings of the 30th International Conference on Machine Learning, ICML 2013</conf-name>, <conf-loc>Atlanta, GA, USA</conf-loc>, <conf-date>16-21 June 2013</conf-date> (<publisher-name>JMLR.org), vol. 28 of JMLR Workshop and Conference Proceedings</publisher-name>), <fpage>1058</fpage>&#x2013;<lpage>1066</lpage>.</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A two-stream network with complementary feature fusion for pest image classification</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>124</volume>, <fpage>106563</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2023.106563</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Lai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>IP102: a large-scale benchmark dataset for insect pest recognition</article-title>,&#x201d; in <conf-name>IEEE Conference on Computer Vision and Pattern Recognition, CVPR 2019</conf-name>, <conf-loc>Long Beach, CA, USA</conf-loc>, <conf-date>June 16-20, 2019</conf-date> (<publisher-name>Computer Vision Foundation/IEEE</publisher-name>), <fpage>8787</fpage>&#x2013;<lpage>8796</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2019.00899</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Convolutional rebalancing network for the classification of large imbalanced rice pest and disease datasets in the field</article-title>. <source>Front. Plant Sci.</source> <volume>12</volume>, <fpage>671134</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2021.671134</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Oh</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Choe</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Cutmix: regularization strategy to train strong classifiers with localizable features</article-title>,&#x201d; in <conf-name>2019 IEEE/CVF International Conference on Computer Vision, ICCV 2019</conf-name>, <conf-loc>Seoul, Korea (South)</conf-loc>, <conf-date>October 27 - November 2, 2019 (IEEE)</conf-date>, <fpage>6022</fpage>&#x2013;<lpage>6031</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV.2019.00612</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ciss&#xe9;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dauphin</surname>
<given-names>Y. N.</given-names>
</name>
<name>
<surname>Lopez-Paz</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>mixup: beyond empirical risk minimization</article-title>,&#x201d; in <conf-name>6th International Conference on Learning Representations, ICLR 2018</conf-name>, <conf-loc>Vancouver, BC, Canada</conf-loc>, <conf-date>April 30 - May 3, 2018</conf-date> (<publisher-name>OpenReview.net</publisher-name>).</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Classification and identification of apple leaf diseases and insect pests based on improved resnet-50 model</article-title>. <source>Horticulturae</source> <volume>9</volume>, <fpage>1046</fpage>. <pub-id pub-id-type="doi">10.3390/horticulturae9091046</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>de Rijke</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Pre-train, interact, fine-tune: a novel interaction representation for text classification</article-title>. <source>Inf. Process. Manag.</source> <volume>57</volume>, <fpage>102215</fpage>. <pub-id pub-id-type="doi">10.1016/j.ipm.2020.102215</pub-id>
</citation>
</ref>
<ref id="B65">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Improving calibration for long-tailed recognition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>, <fpage>16489</fpage>&#x2013;<lpage>16498</lpage>.</citation>
</ref>
<ref id="B66">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>X.-S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.-M.</given-names>
</name>
</person-group> (<year>2020b</year>). &#x201c;<article-title>Bbn: bilateral-branch network with cumulative learning for long-tailed visual recognition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>, <fpage>9719</fpage>&#x2013;<lpage>9728</lpage>.</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Data augmentation in emotion classification using generative adversarial networks</article-title>. <source>Corr. abs/1711</source>, <fpage>00648</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1711.00648</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>