<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1598534</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A multi-scale detection model for tomato leaf diseases with small target detection head</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Sun</surname>
<given-names>Hao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3013129/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Xiaofeng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Xiaofang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Xuewei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cheng</surname>
<given-names>Zhenqi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3010821/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Al-Absi</surname>
<given-names>Mohammed Abdulhakim</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Xiao</surname>
<given-names>Liqun</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Fu</surname>
<given-names>Rui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Shandong Facility Horticulture Bioengineering Research Center, Weifang University of Science and Technology</institution>, <addr-line>Weifang</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Smart Computing, Kyungdong University</institution>, <addr-line>Goseong-gun, Gangwon-do</addr-line>, <country>Republic of Korea</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Computer, Sichuan Technology and Business University</institution>, <addr-line>Chengdu</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Ye Liu, Nanjing Agricultural University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Anjan Debnath, Khulna University of Engineering &amp; Technology, Bangladesh</p>
<p>Karthika J., Sathyabama Institute of Science and Technology, India</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Liqun Xiao, <email xlink:href="mailto:xiaoliqun@my.swjtu.edu.cn">xiaoliqun@my.swjtu.edu.cn</email>; Rui Fu, <email xlink:href="mailto:furui19891209@wfust.edu.cn">furui19891209@wfust.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>16</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1598534</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>03</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>05</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Sun, Li, Li, Wang, Cheng, Al-Absi, Xiao and Fu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Sun, Li, Li, Wang, Cheng, Al-Absi, Xiao and Fu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>In tomato cultivation, various diseases significantly impact tomato quality and yield. The substantial scale differences among diseased leaf targets pose precise detection and identification challenges. Moreover, early detection of disease infection in small leaves during the initial growth stages is crucial for implementing timely intervention and prevention strategies. To address these challenges, we propose a novel tomato disease detection method called TomatoLeafDet, which integrates multi-scale feature processing techniques and small object detection technologies.Initially, we designed a Cross Stage Partial -Serial Multi-kernel Feature Aggregation (CSP-SMKFA) module to extract feature information from targets at different scales, enhancing the model's perception of multi-scale objects. Next, we introduced a Symmetrical Re-calibration Aggregation (SRCA) module, incorporating a bidirectional fusion mechanism between highresolution and low-resolution features. This approach facilitates more comprehensive information transmission between features, further improving the efficacy of multi-scale feature fusion. Finally, we proposed a Re-Calibration Feature Pyramid Network with a small object detection head to consolidate the multi-scale features extracted by the backbone network. This network provides richer multi-scale feature information input for detection heads at various scales. Results indicate that our method outperforms YOLOv9 and YOLOv10 on two datasets. Notably, on the CCMT tomato dataset, the proposed model achieved improvements in mean Average Precision (mAP50) of 4.4%, 1.9%, and 2.3% compared to the baseline model, YOLOv9s, and YOLOv10n, respectively, exhibiting significant efficacy.</p>
</abstract>
<kwd-group>
<kwd>tomato disease detection</kwd>
<kwd>deep learning</kwd>
<kwd>multi-scale detection</kwd>
<kwd>serial multi-kernel feature aggregation</kwd>
<kwd>symmetrical re-calibration aggregation</kwd>
<kwd>FPN</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="4"/>
<equation-count count="14"/>
<ref-count count="30"/>
<page-count count="12"/>
<word-count count="4280"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Tomatoes are one of the most widely consumed vegetables globally, and they are esteemed for their versatility and high nutritional value. Rich in vitamins, minerals, and antioxidants, tomatoes play a crucial role in the balanced diet of most households. However, in recent years, tomato cultivation has faced significant challenges due to increasingly unstable environmental conditions resulting from climate change. These unpredictable weather patterns have led to a surge in various leaf diseases and pest infestations, particularly during the early stages of tomato plant growth <xref ref-type="bibr" rid="B19">Redmond et&#xa0;al. (2018)</xref>. Common diseases such as leaf spot, leaf blight, and yellow leaf curl have become more prevalent, severely impacting both the yield and quality of tomato crops <xref ref-type="bibr" rid="B10">Liu and Wang (2021)</xref>. The rising incidence of diseases has prompted farmers to increase their reliance on pesticides as a defensive measure. However, this approach has also brought about a series of problems. For instance, excessive use of pesticides may lead to the development of pesticide resistance in tomato plants, necessitating the application of different and more potent, expensive pesticides. This vicious cycle increases the economic burden on farmers and poses a serious threat to ecosystems. Furthermore, pesticide residues on tomato surfaces may significantly reduce the nutritional value of the fruit and potentially endanger consumer health <xref ref-type="bibr" rid="B1">Amr and Raie (2022)</xref>.</p>
<p>Early intervention and prevention strategies are crucial for mitigating these issues. Farmers can significantly reduce pesticide use, minimize expenses, and ensure better crop yield and quality by detecting and addressing diseases initially. This approach protects the nutritional integrity of tomatoes and enhances the economic benefits of tomato cultivation. However, implementing effective early intervention strategies requires accurate and timely leaf disease detection methods, which have proven to be a formidable challenge in agricultural practices, as illustrated in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. Traditional manual inspection methods for identifying tomato leaf diseases have numerous limitations; these methods are labor-intensive, time-consuming, and often inefficient <xref ref-type="bibr" rid="B21">Shoaib et&#xa0;al. (2023)</xref>. Furthermore, the subtle symptoms of early-stage diseases can be easily overlooked even by experienced professionals, leading to delayed interventions and increased vegetable losses <xref ref-type="bibr" rid="B3">Demilie (2024)</xref>. The need for more advanced, reliable, and efficient detection methods has become increasingly evident in modern agriculture.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The current status of tomato cultivation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g001.tif"/>
</fig>
<p>The emergence of deep learning techniques has opened a new phase for vegetable disease detection methods. Convolutional Neural Networks (CNNs) have provided promising solutions for vegetable disease classification. Various CNN-based models have been proposed and applied in agricultural environments, demonstrating significant advantages over traditional methods. However, these initial deep learning approaches primarily focused on classification tasks <xref ref-type="bibr" rid="B4">Ferdinand and Al Maki (2022)</xref>; <xref ref-type="bibr" rid="B14">Madhav et&#xa0;al. (2021)</xref>, failing to address the crucial issue of disease localization. This limitation hindered their ability to fully replace manual inspection processes, as precise location information is essential for targeted treatment and intervention strategies. The latest advances in computer vision have facilitated the development of faster and more accurate object detection models. These models can classify diseases and precisely locate the diseased areas in images, marking a significant leap in automated plant disease detection. Object detection models can be broadly categorized into two types: two-stage detectors and single-stage detectors. The Faster R-CNN <xref ref-type="bibr" rid="B20">Ren et&#xa0;al. (2015)</xref> model is particularly prominent among two-stage models. The methods divide the task of identifying objects into two key steps: the first step involves producing potential areas or bounding boxes that might contain objects of interest; the second step focuses on categorizing these candidate regions and determining their exact positions <xref ref-type="bibr" rid="B6">He et&#xa0;al. (2023)</xref>; <xref ref-type="bibr" rid="B30">Zhao et&#xa0;al. (2022)</xref>; <xref ref-type="bibr" rid="B25">Wang et&#xa0;al. (2022b)</xref>. In contrast, single-stage detectors, especially the YOLO (You Only Look Once) <xref ref-type="bibr" rid="B16">Redmon et&#xa0;al. (2016)</xref>; <xref ref-type="bibr" rid="B17">Redmon and Farhadi (2016)</xref>; <xref ref-type="bibr" rid="B18">Redmon and Farhadi (2018)</xref>; <xref ref-type="bibr" rid="B2">Bochkovskiy et&#xa0;al. (2020)</xref>; <xref ref-type="bibr" rid="B23">Ultralytics (2022)</xref>; <xref ref-type="bibr" rid="B24">Wang et&#xa0;al. (2022a)</xref>; <xref ref-type="bibr" rid="B28">Wang et&#xa0;al. (2024)</xref>; <xref ref-type="bibr" rid="B9">Li et&#xa0;al. (2022)</xref>; <xref ref-type="bibr" rid="B8">Jocher et&#xa0;al. (2023)</xref> series models, have garnered attention due to their efficiency and accuracy. Researchers have also begun to focus on improving the performance of YOLO models in vegetable disease detection. For instance: <xref ref-type="bibr" rid="B27">Wang et&#xa0;al. (2021)</xref> proposed a YOLOv3 detection model incorporating a novel IOU by enhancing tomato pest and disease sample data <xref ref-type="bibr" rid="B27">Wang et&#xa0;al. (2021)</xref>. <xref ref-type="bibr" rid="B12">Liu et&#xa0;al. (2022)</xref> proposed a novel YOLOv4 detection model for tomato pests by integrating three sets of attention mechanisms <xref ref-type="bibr" rid="B12">Liu et&#xa0;al. (2022)</xref>. <xref ref-type="bibr" rid="B13">Liu et&#xa0;al. (2023)</xref> developed a YOLOv5 model with a novel loss function to address the detection of tomato brown rot <xref ref-type="bibr" rid="B13">Liu et&#xa0;al. (2023)</xref>. <xref ref-type="bibr" rid="B28">Wang et&#xa0;al. (2024)</xref> integrated Transformer into YOLOv8 to improve tomato disease detection, enhancing the model&#x2019;s ability to capture disease detail features <xref ref-type="bibr" rid="B26">Wang and Liu (2024)</xref>. <xref ref-type="bibr" rid="B7">Jiang et&#xa0;al. (2024)</xref> combined Swin Transformer and CNN to optimize YOLOv8&#x2019;s feature extraction capability, improving the model&#x2019;s detection ability for cabbage diseases in complex environments <xref ref-type="bibr" rid="B7">Jiang et&#xa0;al. (2024)</xref>. <xref ref-type="bibr" rid="B11">Liu and Wang (2024)</xref> proposed a multi-source information fusion method based on YOLOv8 to improve the accuracy of multi-vegetable disease detection <xref ref-type="bibr" rid="B11">Liu and Wang (2024)</xref>.</p>
<p>Although these improved YOLO models address some issues in crop disease detection, they primarily focus on solving problems such as small sample sizes, complex background environments, and model parameter optimization. However, the impact of leaf size variation on detection accuracy has rarely been addressed or mentioned. As shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, some data samples are collected from one or two leaves, while others are collected from multiple leaves. Despite having the same sample size, the scale of the target data varies significantly. Most existing models have been optimized to detect diseases on leaves of normal size, potentially overlooking early infections on smaller leaves. This limitation is particularly important in the context of early intervention strategies. Therefore, it is crucial to be able to specifically perceive diseases on leaves of various sizes, especially smaller ones. To address this critical gap in current tomato disease leaf detection methods, we propose a novel tomato disease leaf detection algorithm, TomatoLeafDet. The contributions of this study can be summarized as follows:</p>
<list list-type="bullet">
<list-item>
<p>We propose a real-time detection model named &#x201c;TomatoLeafDet&#x201d;. By integrating various novel multiscale strategies, this model aims to simultaneously capture and process features of leaves of various sizes, with particular emphasis on improving the detection capability for smaller leaves.</p>
</list-item>
<list-item>
<p>We propose a Cross Stage Partial - Serial Multi-kernel Feature Aggregation (CSP-SMKFA) module, which introduces a serial multi-kernel convolutional network capable of extracting multi-scale feature information from the input. It combines 1x1 convolutional layers and residual connections to fuse features of different scales, enhancing the model&#x2019;s expressive ability.</p>
</list-item>
<list-item>
<p>We propose a Symmetrical Re-calibration Aggregation (SRCA) module. Through an adaptive attention mechanism, it adaptively adjusts the weights of features according to the different resolutions and contents of feature maps, thereby better capturing the multi-scale features of the target.</p>
</list-item>
<list-item>
<p>Utilizing the CSP-SMKFA and SRCA modules, we propose a Re-Calibration Feature Pyramid Network (Re-CalibrationFPN) with a small target detection head. It introduces a bidirectional fusion mechanism between high-resolution and low-resolution features, enabling more comprehensive information transmission between features and further enhancing the effect of multi-scale feature fusion. Simultaneously, it incorporates a specialized small target detection head aimed at improving the sensitivity and accuracy of disease detection on smaller leaves.</p>
</list-item>
<list-item>
<p>Through comparative and ablation experiments, our model demonstrates significant advantages compared to the baseline model, while also surpassing YOLOv9 and YOLOv10. Our method has the potential to provide a more comprehensive and detailed tool for early disease intervention in tomato cultivation.</p>
</list-item>
</list>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Sample Data Collected from Tomato Leaves at Different Scales.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g002.tif"/>
</fig>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Materials</title>
<sec id="s2_1_1">
<label>2.1.1</label>
<title>Dataset</title>
<p>We evaluated our model on two public datasets: the PlantDoc plant disease detection dataset <xref ref-type="bibr" rid="B22">Singh et&#xa0;al. (2019)</xref> and the CCMT tomato disease dataset <xref ref-type="bibr" rid="B15">Mensah et&#xa0;al. (2023)</xref>. The PlantDoc contains 2,598 data samples for 13 plant species and 17 disease classes. The CCMT consists of 4960 samples distributed across 5 tomato leaf disease classes. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> presents the details of specific diseased leaves: (a) septoria leaf spot, (b) leaf blight, (c) leaf verticillium wilt, (d) healthy leaf, (e) leaf curl.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Display of five tomato disease leaves. <bold>(a)</bold> septoria leaf spot, <bold>(b)</bold> leaf blight, <bold>(c)</bold> leaf verticillium wilt, <bold>(d)</bold> healthy leaf, <bold>(e)</bold> leaf curl.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g003.tif"/>
</fig>
<p>However, the detection targets in this tomato data sample were not annotated. Therefore, we first used the LabelImg annotation tool to annotate the 4960 images in the tomato leaf disease dataset, as shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. Then, corresponding XML files containing bounding box coordinates and category labels were generated. Subsequently, these XML files were converted into txt annotation files used by the YOLO model. After preprocessing, the dataset was partitioned into three distinct subsets: 4059 instances for training, 451 for validation, and 450 for testing. The training set served to fit the model parameters, while the validation set was employed to fine-tune hyperparameters during the learning process and conduct initial assessments of model performance. Ultimately, the test set was utilized to gauge the final model&#x2019;s capacity for generalization.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>LabelImg annotation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g004.tif"/>
</fig>
</sec>
<sec id="s2_1_2">
<label>2.1.2</label>
<title>Implementation details</title>
<p>The models proposed in the study use PyTorch as the learning framework, and the experimental acceleration is configured with AMD and RTX 4090. During the training process, 640 &#xd7; 640 images are randomly cropped from the sample images. The batch size for each GPU is set to 12. Each model undergoes a training regimen consisting of 150 iterations. The optimization process employs the Adam algorithm, with the learning rate initially set at 0.01. As shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, detailed experimental setup details are presented.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Our experimental environment.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Computational Ecosystem</th>
<th valign="top" align="center">Details</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">System software</td>
<td valign="top" align="center">64 Bit Windows 11</td>
</tr>
<tr>
<td valign="top" align="center">Coding</td>
<td valign="top" align="center">Python 3.9</td>
</tr>
<tr>
<td valign="top" align="center">GPU</td>
<td valign="top" align="center">RTX 4090</td>
</tr>
<tr>
<td valign="top" align="center">CPU</td>
<td valign="top" align="center">4.20 GHz AMD 16 Core Processor</td>
</tr>
<tr>
<td valign="top" align="center">Pytorch</td>
<td valign="top" align="center">2.2.2</td>
</tr>
<tr>
<td valign="top" align="center">CUDA</td>
<td valign="top" align="center">11.8</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Methods</title>
<sec id="s2_2_1">
<label>2.2.1</label>
<title>Macroscopic architecture</title>
<p>
<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> illustrates the structural diagram of the TomatoLeafDet model architecture for detecting tomato leaf diseases. It is primarily divided into three parts: Backbone, Re-Calibration FPN, and P2345 Head. Through the introduction of new modules and optimized design, this model aims to enhance multi-scale feature extraction, fusion, and detection capabilities.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>The overall architecture of TomatoLeafDet.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g005.tif"/>
</fig>
<p>In the initial step of the Backbone section, the model replaces the original C2F module with CSPSMKFA, combined with Conv convolutional modules for feature extraction. Through multi-scale partial convolution, the CSP-SMKFA module can more effectively capture different scale information in the image, especially enhancing feature expression ability when dealing with large differences in target sizes. This improvement helps to enhance the model&#x2019;s capture of input image features, providing richer semantic information for subsequent feature fusion processing. Simultaneously, the SPPF module is retained at the end of the Backbone, further processing features at multiple scales, and increasing the robustness of feature representation. Subsequently, in the Neck section, the model introduces a new Re-Calibration FPN, where the SRCA module can perform spatial recalibration on feature maps at different levels, enabling the network to more effectively select and focus on useful feature information. Combined with the CSP-SMKFA module, it enhances the effect of multi-scale feature fusion. This structure allows for more comprehensive information transmission between features at different levels, optimizing the detection performance of multi-scale targets. Finally, in the Head section, the model introduces a P2 detection head specifically for small target detection, working in conjunction with P3, P4, and P5 detection heads. This design can better capture fine-grained information in low-level feature maps, thereby improving the model&#x2019;s detection accuracy for small-scale targets. In terms of loss functions, the model still employs traditional regression loss (box loss), classification loss (cls loss), and DFL loss (dfl loss) to ensure the accuracy of predicted bounding box positions and categories. Overall, the model has been optimized at the Backbone, Neck, and Head levels, particularly through the introduction of CSP-SMKFA and SRCA modules, enhancing the ability of multi-scale feature extraction and fusion. This makes the improved TomatoLeafDet model more robust and accurate in multi-scale object detection tasks.</p>
</sec>
<sec id="s2_2_2">
<label>2.2.2</label>
<title>Cross stage partial - serial multi-kernel feature aggregation</title>
<p>The CSP-SMKFA module adopts a novel partial multi-scale feature concatenation aggregation strategy to enhance computational efficiency and feature representation capability. As shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>, its structural process is as follows: Initially, the input features undergo preliminary processing through Conv1 x 1, subsequently dividing into two branches. The main branch then employs a cascaded convolution structure: a) Conv3 x 3 extracts local features, b) Conv5 x 5 captures medium-scale contextual information, and c) Conv7 x 7 obtains global features with a larger receptive field. Each convolutional layer allocates 50% of its output channels to the next convolutional layer and skips the connection. The skip connection branch directly transmits the remaining 50% of the Conv1 x 1, Conv3 x 3, Conv5 x 5, and Conv7 x 7 outputs, preserving original feature information. A Concat operation then concatenates features of different scales (1x1, 3x3, 5x5, 7x7 convolution outputs) along the channel dimension. Subsequently, Conv1 x 1 further fuses these multi-scale features, producing a unified feature representation. Finally, a residual connection adds the original input features to the processed features, forming the final output. This design effectively integrates spatial information at different scales while reducing computational complexity through partial channel multi-scale processing. The cascaded convolution structure progressively expands the receptive field, capturing multi-scale context. Skip connections and residual connections ensure the preservation of original information, facilitating gradient propagation. The final feature aggregation and fusion steps further enhance the model&#x2019;s expressive capacity, enabling it to better adapt to the demands of various multi-scale object detection tasks. Therefore, the process of CSP-SMKFA can be formulated as:</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>The structure of CSP-SMKFA and SRCA.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g006.tif"/>
</fig>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>S</mml:mtext>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>S</mml:mtext>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mtext>S</mml:mtext>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mn>7</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mtext>Output</mml:mtext>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>Concat</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mn>5</mml:mn>
</mml:msub>
<mml:msup>
<mml:mn>2</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>O</mml:mi>
<mml:mn>7</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2295;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mtext>S</mml:mtext>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents splitting the output feature map vector along the channel dimension with a ratio of 0.5 for each part. &#x2295; represents addition. Concat(&#xb7;) is concatenation.</p>
</sec>
<sec id="s2_2_3">
<label>2.2.3</label>
<title>Symmetrical re-calibration aggregation</title>
<p>When processing targets of different scales, there is a tendency to lose significant semantic information. Shallow features contain less semantic content but have clear boundaries and rich details. Deep features, however, encompass abundant semantic information. Directly fusing shallow and deep features may result in redundant information in the fused features <xref ref-type="bibr" rid="B5">Han et&#xa0;al. (2019)</xref>. To address this issue, we propose the Symmetrical Re-calibration Aggregation (SRCA) module, which utilizes a self-attention mechanism to fuse high-resolution and low-resolution features. Feature fusion is conducted selectively through weighting, capturing more comprehensive fine-grained information, and recalibrating target positions.</p>
<p>
<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> illustrates the structure of the SRCA module, designed to effectively fuse high-resolution and low-resolution features. The module comprises four parallel complementary processing paths, each handling high-resolution (H1) and low-resolution (L1) features. For instance, in the first two paths&#x2019; processing flow: Initially, high-resolution feature H1 is processed through a 1x1 convolutional layer and Convolutional Block Attention Module (CBAM) <xref ref-type="bibr" rid="B29">Woo et&#xa0;al. (2018)</xref>, yielding compressed mapped feature H1&#x2019; and attention weight a1. Low-resolution feature L1 undergoes the same processing, producing L1&#x2019;. Subsequently, through a self-attention mechanism, (1-a1) is fused with L1 and L1&#x2019;. Then, the fused features are added to the original H1. The latter two paths follow a similar processing flow but interchange the input positions of high and low-resolution features: First, L1 is processed through 1x1 convolution and CBAM, yielding L1&#x2019; and attention weight b1. Then, H1 undergoes the same processing to obtain H1&#x2019;. Internal upsampling and downsampling functions adjust the sizes of (1-b1), H1, and H1&#x2019; for three-part fusion. The fusion result is then added to the original L1. Finally, the outputs of these four paths are channel-fused and passed through a Conv3 x 3 layer to output a rich fused feature. This design allows the model to adaptively select and fuse features from different resolutions, effectively capturing multi-scale information. By cross-processing high and low-resolution features, the module can more comprehensively utilize spatial information at different scales, thereby enhancing the richness and effectiveness of feature representation. Therefore, the process of SRCA can be formulated as:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>CC</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msubsup>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mtext>CC</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>output</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2295;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2295;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2299;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>output</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2295;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2295;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>'</mml:mo>
</mml:msubsup>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2299;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mtext>Output</mml:mtext>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>Conv</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>Concat</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>output</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mtext>output</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>w</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where CC represents Conv<sub>1&#xd7;1</sub> and CBAM, used to process input features, halve the channels, and obtain new feature maps and weight coefficients. &#x2295; represents addition. &#x2299; represents matrix product. Concat(&#xb7;) is concatenation.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Experiments</title>
<sec id="s3_1">
<label>3.1</label>
<title>Experimental indicators</title>
<p>In this study, we employed multiple metrics to evaluate our model&#x2019;s performance: Parameters, GFLOPs (Giga Floating Point Operations Per Second), mean Average Precision (mAP50-90), and mean Average Precision (mAP50). Among these, mAP50 was selected as the primary evaluation metric. The calculation process for mean Average Precision is delineated in <xref ref-type="disp-formula" rid="eq11">Equations 11</xref>-<xref ref-type="disp-formula" rid="eq14">14</xref>.</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x222b;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The variable K signifies the total count of distinct object classifications within the dataset, while each class&#x2019;s precision is quantified by its specific Average Precision (AP) score. In the performance evaluation equations, several key indicators are utilized: True Positives (TP) represent accurately identified instances of the target condition, False Positives (FP) indicate cases where the algorithm incorrectly flagged non-existent conditions as present, and False Negatives (FN) encompass actual occurrences of the condition that the system failed to recognize.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Comparison studies</title>
<p>To validate the effectiveness of the proposed model in this study, we first conducted comparative experiments on the publicly available PlantDoc dataset. We compared current mainstream object detection models, with each comparative model using the same experimental parameters. <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> presents the experimental results of our proposed TomatoLeafDet model compared to other state-of-the-art real-time detection models on the PlantDoc dataset. Compared to the baseline model, TomatoLeafDet achieved improvements of 9.3% and 13.7% in mAP50 and mAP50-95, respectively. When compared to the advanced YOLOv10n, TomatoLeafDet also demonstrated a 7.2% increase in mAP50 and a 12.2% increase in mAP5095. <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> visually illustrates the performance gap between the TomatoLeafDet model and the baseline model, clearly showing that TomatoLeafDet consistently outperforms the baseline model throughout the entire training process.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Comparison with advanced real-time object frameworks on the PlantDoc dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Model</th>
<th valign="top" align="center">Parameters</th>
<th valign="top" align="center">GFLOPs</th>
<th valign="top" align="center">mAP50-95</th>
<th valign="top" align="center">mAP50</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">YOLOv8n (baseline)</td>
<td valign="top" align="center">3,011,498</td>
<td valign="top" align="center">8.1</td>
<td valign="top" align="center">0.282</td>
<td valign="top" align="center">0.420</td>
</tr>
<tr>
<td valign="top" align="center">YOLOv9t</td>
<td valign="top" align="center">2,628,260</td>
<td valign="top" align="center">10.7</td>
<td valign="top" align="center">0.302</td>
<td valign="top" align="center">0.443</td>
</tr>
<tr>
<td valign="top" align="center">YOLOv10n</td>
<td valign="top" align="center">2,706,116</td>
<td valign="top" align="center">8.3</td>
<td valign="top" align="center">0.286</td>
<td valign="top" align="center">0.428</td>
</tr>
<tr>
<td valign="top" align="center">Ours</td>
<td valign="top" align="center">3,351,832</td>
<td valign="top" align="center">9.8</td>
<td valign="top" align="center">0.321</td>
<td valign="top" align="center">0.459</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>The mAP comparison between baseline and TomatoLeafDet on PlantDoc dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g007.tif"/>
</fig>
<p>After that, we conducted a comprehensive comparative evaluation of the TomatoLeafDet model on the tomato leaf disease dataset. As shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, we selected the two-stage model Faster R-CNN and current mainstream single-stage YOLO series models for comparison. Notably, compared to these real-time detection models, our model demonstrated superior performance in mAP50&#x2013;95 and mAP50, primarily attributed to the CSP-SMKFA and Re-CalibrationFPN multi-scale feature processing styles, which enhanced the model&#x2019;s multi-scale perception capabilities. Compared to the baseline model, TomatoLeafDet achieved improvements of 4.5% and 4.4% in mAP50&#x2013;95 and mAP50, respectively. Compared to the two stage Faster r-cnn model, TomatoLeafDet significantly reduced parameters and GFLOPs while improving mAP50 by 6.1%. Our model also showed significant advantages over larger single-stage models, for example, compared to YOLOv8s, TomatoLeafDet reduced parameters by approximately 70% while notably increasing mAP50 by 1.5%. Compared to YOLOv9s, TomatoLeafDet reduced GFLOPs by about 74.7% while significantly improving mAP50 by 2.0%. It is also worth noting that compared to the current advanced model YOLOv10n, our TomatoLeafDet model surpassed it by 0.021 mAP50. <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref> visually illustrates the average precision advantage of our model over the baseline model throughout the entire training process.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Object detection with different frameworks on CCMT tomato dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Model</th>
<th valign="top" align="center">Parameters</th>
<th valign="top" align="center">GFLOPs</th>
<th valign="top" align="center">mAP50-95</th>
<th valign="top" align="center">mAP50</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Faster-rcnn</td>
<td valign="top" align="center">137,101,141</td>
<td valign="top" align="center">370.0</td>
<td valign="top" align="center">0.670</td>
<td valign="top" align="center">0.888</td>
</tr>
<tr>
<td valign="top" align="center">YOLO7</td>
<td valign="top" align="center">3,760,872</td>
<td valign="top" align="center">105.5</td>
<td valign="top" align="center">0.641</td>
<td valign="top" align="center">0.853</td>
</tr>
<tr>
<td valign="top" align="center">YOLOv8n (baseline)</td>
<td valign="top" align="center">3,011,823</td>
<td valign="top" align="center">8.2</td>
<td valign="top" align="center">0.735</td>
<td valign="top" align="center">0.905</td>
</tr>
<tr>
<td valign="top" align="center">YOLOv8s</td>
<td valign="top" align="center">11,127,519</td>
<td valign="top" align="center">28.4</td>
<td valign="top" align="center">0.757</td>
<td valign="top" align="center">0.931</td>
</tr>
<tr>
<td valign="top" align="center">YOLOv9t</td>
<td valign="top" align="center">2,618,510</td>
<td valign="top" align="center">10.7</td>
<td valign="top" align="center">0.739</td>
<td valign="top" align="center">0.913</td>
</tr>
<tr>
<td valign="top" align="center">YOLOv9s</td>
<td valign="top" align="center">9,601,118</td>
<td valign="top" align="center">38.7</td>
<td valign="top" align="center">0.760</td>
<td valign="top" align="center">0.927</td>
</tr>
<tr>
<td valign="top" align="center">YOLOv10n</td>
<td valign="top" align="center">2,696,366</td>
<td valign="top" align="center">8.2</td>
<td valign="top" align="center">0.743</td>
<td valign="top" align="center">0.924</td>
</tr>
<tr>
<td valign="top" align="center">Ours</td>
<td valign="top" align="center">3,348,532</td>
<td valign="top" align="center">9.8</td>
<td valign="top" align="center">0.768</td>
<td valign="top" align="center">0.945</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>The mAP comparison between baseline and TomatoLeafDet on the CCMT tomato dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g008.tif"/>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Ablation studies</title>
<p>TomatoLeafDet incorporates Re-CalibrationFPN, CSP-SMKFA, and P2 detection head. We conducted ablation experiments on these components sequentially, with results shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>. We first added the ReCalibrationFPN without the P2 head structure, which improved performance by 1.7% mAP. Subsequently, we added the P2 detection head on this basis, further increasing the mean Average Precision (mAP) by 1.1%. To verify the effectiveness of the CSP-SMKFA module, we independently added it to the baseline model, resulting in a 3.0% improvement in mAP50&#x2013;90 and a 2.7% improvement in mAP50. Finally, we integrated both Re-CalibrationFPN with P2 and CSP-SMKFA into the baseline model, constructing our new model, TomatoLeafDet. As shown in the last row of <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, the combination of our proposed modules effectively improved the model&#x2019;s performance by 4.5% and 4.4% in mAP50&#x2013;95 and mAP50, respectively. This validates the effectiveness of our proposed Re-CalibrationFPN, CSP-SMKFA, and P2 detection head in tomato leaf disease detection.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Ablation studies of key components on CCMT tomato dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Model</th>
<th valign="top" align="center">Re-Calibration FPN without P2</th>
<th valign="top" align="center">Re-Calibration FPN with P2</th>
<th valign="top" align="center">CSP-SMKFA</th>
<th valign="top" align="center">Parameters</th>
<th valign="middle" align="center">GFLOPs</th>
<th valign="middle" align="center">mAP50-95</th>
<th valign="middle" align="center">mAP50</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">1</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="bottom" align="center">3,011,823</td>
<td valign="bottom" align="center">8.2</td>
<td valign="bottom" align="center">0.735</td>
<td valign="bottom" align="center">0.905</td>
</tr>
<tr>
<td valign="top" align="center">2</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">3,806,303</td>
<td valign="top" align="center">9.9</td>
<td valign="top" align="center">0.743</td>
<td valign="top" align="center">0.920</td>
</tr>
<tr>
<td valign="top" align="center">3</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">3,761,476</td>
<td valign="top" align="center">10.9</td>
<td valign="top" align="center">0.746</td>
<td valign="top" align="center">0.930</td>
</tr>
<tr>
<td valign="top" align="center">4</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">2,802,695</td>
<td valign="top" align="center">8.1</td>
<td valign="top" align="center">0.757</td>
<td valign="top" align="center">0.929</td>
</tr>
<tr>
<td valign="top" align="center">5</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">3,402,375</td>
<td valign="top" align="center">8.9</td>
<td valign="top" align="center">0.760</td>
<td valign="top" align="center">0.936</td>
</tr>
<tr>
<td valign="top" align="center">Ours</td>
<td valign="top" align="center">&#xd7;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">3,348,532</td>
<td valign="top" align="center">9.8</td>
<td valign="top" align="center">0.768</td>
<td valign="top" align="center">0.945</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The symbol &#x201c;&#x221a;&#x201d; indicates that the module is included, while the symbol &#x201c;&#xd7;&#x201d; indicates that the module is removed.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Visual comparative studies</title>
<p>To visualize the advantages of TomatoLeafDet for different sizes of tomato leaves detection, the differentiated detection results of the baseline model, the advanced model yolov10n, and our proposed model on the CCMT tomato dataset are shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>. It can be observed that our model exhibits increased attention and sensitivity to small-scale leaves while simultaneously detecting normal-sized leaves. In contrast, the other two models predominantly focus on normal-scale target leaves. Consequently, these results demonstrate that our proposed novel model performs excellently in tomato leaf disease detection. Therefore, it provides a potential solution for early predicting and preventing tomato leaf diseases.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Comparison of the visualization detection result of different models, <bold>(a)</bold> Baseline Model; <bold>(b)</bold> Yolov10n Model; <bold>(c)</bold> TomatoLeafDet.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1598534-g009.tif"/>
</fig>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusion</title>
<p>In this study, we introduced Re-CalibrationFPN as a solution to address the limitations of the baseline model in handling multi-scale object features. It comprises two key components: CSP-SMKFA and SRCA. CSP-SMKFA utilizes concatenated multi-kernel partial convolutions to perceive multi-scale feature information. At the same time, SRCA further effectively integrates deep rich information and shallow detail information through a symmetrically complementary structure. Additionally, we incorporated a P2 small object detection head, enabling the model to focus on detailed features of small-sized objects while attending to multi-scale object features. Through extensive experimentation, we demonstrated that our proposed model achieved advanced performance on both the PlantDoc dataset and the CCMT tomato dataset, surpassing current mainstream object detection models.</p>
<p>Although our model has significantly advanced tomato leaf disease detection, further research is required to bridge the gap between experimental results and practical applications. In subsequent studies, we plan to collect and establish more comprehensive datasets to train and improve model performance. We will continue to address the limitations in multi-scale and small object detection while gradually considering research into lightweight models. Future research focuses on integrating lightweight detection models with drones and robots, accelerating the establishment of early prediction and prevention systems for precision vegetable cultivation.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s6" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>Written informed consent was obtained from the individual(s) for the publication of any potentially identifiable images or data included in this article.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>HS: Conceptualization, Methodology, Investigation, Data curation, Formal analysis, Writing &#x2013; original draft. XFeL: Conceptualization, Methodology, Formal analysis, Writing &#x2013; review &amp; editing. XFaL: Writing &#x2013; review &amp; editing. XW: Writing &#x2013; review &amp; editing. ZC: Investigation, Data curation, Resources, Writing &#x2013; review &amp; editing. MA-A: Formal analysis, Writing &#x2013; review &amp; editing. LX: Writing &#x2013; review &amp; editing. RF: Supervision, Conceptualization, Methodology, Formal analysis, Funding acquisition, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. Shandong Provincial Natural Science Foundation (Grant No. ZR2025QC649), Weifang University of Science and Technology A-Class Doctoral Research Fund (Grant No. KJRC2024006).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>The authors would like to acknowledge the contributions of the participants in this study.</p>
</ack>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2025.1598534/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2025.1598534/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Image1.jpeg" id="SM1" mimetype="image/jpeg"/>
<supplementary-material xlink:href="Image2.jpeg" id="SM2" mimetype="image/jpeg"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Amr</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Raie</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Tomato components and quality parameters. a review</article-title>. <source>Jordan J. Agric. Sci.</source> <volume>18</volume>, <fpage>199</fpage>&#x2013;<lpage>220</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.35516/jjas.v18i3.444</pub-id>
</citation></ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Yolov4: Optimal speed and accuracy of object detection</source> (<publisher-loc>Ithaca, New York, USA (Cornell University Library)</publisher-loc>: <publisher-name>ArXiv</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2004.10934</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Demilie</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Plant disease detection and classification techniques: a comparative study of the performances</article-title>. <source>J. Big Data</source> <volume>11</volume>, <fpage>5</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s40537-023-00863-9</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferdinand</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Al Maki</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Broccoli leaf diseases classification using support vector machine with particle swarm optimization based on feature selection</article-title>. <source>Int. J. Adv. Intelligent Inf.</source> <volume>8</volume>, <fpage>337</fpage>&#x2013;<lpage>348</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.26555/ijain.v8i3.951</pub-id>
</citation></ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Ghostnet: More features from cheap operations</source>. (<publisher-loc>Ithaca, New York, USA</publisher-loc>: <publisher-name>ArXiv</publisher-name>).</citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Mfaster r-cnn for maize leaf diseases detection based on machine vision</article-title>. <source>Arab J. Sci. Eng.</source> <volume>48</volume>, <fpage>1437</fpage>&#x2013;<lpage>1449</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s13369-022-06851-0</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Field cabbage detection and positioning system based on improved yolov8n</article-title>. <source>Plant Methods</source> <volume>20</volume>, <fpage>96</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-024-01226-y</pub-id>, PMID: <pub-id pub-id-type="pmid">38902736</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Jocher</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Chaurasia</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Yolo by ultralytics</source>.</citation></ref>
<ref id="B9">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <source>Yolov6: A single-stage object detection framework for industrial applications</source>.</citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Plant diseases and pests detection based on deep learning: a review</article-title>. <source>Plant Methods</source> <volume>17</volume>, <elocation-id>22</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-021-00722-9</pub-id>, PMID: <pub-id pub-id-type="pmid">33627131</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Multisource information fusion method for vegetable disease detection</article-title>. <source>BMC Plant Biol.</source> <volume>24</volume>, <fpage>738</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12870-024-05346-4</pub-id>, PMID: <pub-id pub-id-type="pmid">39095689</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Miao</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Tomato pest recognition algorithm based on improved yolov4</article-title>. <source>Front. Plant Sci.</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2022.814681</pub-id>, PMID: <pub-id pub-id-type="pmid">35909759</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Miao</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Tomato brown rot disease detection using improved yolov5 with attention mechanism</article-title>. <source>Front. Plant Sci.</source> <volume>14</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2023.1289464</pub-id>, PMID: <pub-id pub-id-type="pmid">38053763</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Madhav</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Jyothi</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Kalyani</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Prediction of pesticides and identification of diseases in fruits using support vector machine (svm) and iot</article-title>,&#x201d; in <conf-name>AIP Conference Proceedings</conf-name>, <source>1st International Conference on Advances in Signal Processing, VLSI, Communications and Embedded Systems (ICSVCE-2021)</source> (<publisher-loc>Melville, NY, USA</publisher-loc>: <publisher-name>AIP Publishing</publisher-name>) Vol. <volume>2407</volume>. <fpage>020016</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1063/5.0074595</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mensah</surname> <given-names>P. K.</given-names>
</name>
<name>
<surname>Akoto-Adjepong</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Adu</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Ayidzoe</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Bediako</surname> <given-names>E. A.</given-names>
</name>
<name>
<surname>Nyarko-Boateng</surname> <given-names>O.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <source>Ccmt: Dataset for crop pest and disease detection</source>. (<publisher-loc>Amsterdam, The Netherlands</publisher-loc>: <publisher-name>Elsevier</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.dib.2023.109306</pub-id>, PMID: <pub-id pub-id-type="pmid">37360671</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Divvala</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>You only look once: Unified, real-time object detection</article-title>. <source>ArXiv</source>. <fpage>779</fpage>&#x2013;<lpage>788</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2016.91</pub-id>
</citation></ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Yolo9000: Better, faster, stronger</source> (<publisher-loc>Ithaca, New York, USA (Cornell University Library)</publisher-loc>: <publisher-name>ArXiv</publisher-name>), <page-range>779&#x2013;788</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1612.08242</pub-id>
</citation></ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Yolov3: An incremental improvement</source> (<publisher-loc>Ithaca, New York, USA (Cornell University Library)</publisher-loc>: <publisher-name>ArXiv</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1804.02767</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Redmond</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Thorp</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Desa</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hasfalina</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Taheri</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Review of optimum temperature humidity and vapour pressure deficit for microclimate evaluation and control in greenhouse cultivation of tomato: A review</article-title>. <source>Int. Agrophysics</source> <volume>32</volume>, <fpage>287</fpage>&#x2013;<lpage>302</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1515/intag-2017-0005</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <source>Faster r-cnn: Towards real-time object detection with region proposal networks</source>. (<publisher-loc>Ithaca, New York, USA</publisher-loc>: <publisher-name>ArXiv</publisher-name>).</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shoaib</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Shah</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Ei-Sappagh</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ullah</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Alenezi</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>An advanced deep learning models-based plant disease detection: a review of recent research</article-title>. <source>Front. Plant Sci.</source> <volume>14</volume>, <elocation-id>1158933</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2023.1158933</pub-id>, PMID: <pub-id pub-id-type="pmid">37025141</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Singh</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Jain</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Jain</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Kayal</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Kumawat</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Batra</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Plantdoc: A dataset for visual plant disease detection</source> (<publisher-loc>Ithaca, New York, USA (Cornell University Library)</publisher-loc>: <publisher-name>ArXiv</publisher-name>). doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3371158.3371196</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="web">
<person-group person-group-type="author">
<collab>Ultralytics</collab>
</person-group> (<year>2022</year>). <source>ultralytics/yolov5</source>.</citation></ref>
<ref id="B24">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2022</year>a). <source>Yolov7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</source>.</citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Ling</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Nie</surname> <given-names>L.</given-names>
</name>
<name>
<surname>An</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>b). <article-title>An improved faster r-cnn model for multi-object tomato maturity detection in complex scenarios</article-title>. <source>Ecol. Inf.</source> <volume>72</volume>, <elocation-id>101886</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2022.101886</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An efficient deep learning model for tomato disease detection</article-title>. <source>Plant Methods</source> <volume>20</volume>, <fpage>61</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-024-01188-1</pub-id>, PMID: <pub-id pub-id-type="pmid">38725014</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Early real-time detection algorithm of tomato diseases and pests in the natural environment</article-title>. <source>Plant Methods</source> <volume>17</volume>, <fpage>43</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-021-00745-2</pub-id>, PMID: <pub-id pub-id-type="pmid">33892765</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Yeh</surname> <given-names>I.-H.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Yolov9: Learning what you want to learn using programmable gradient information</source>.</citation></ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Woo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Cbam: Convolutional block attention module</source>. (<publisher-loc>Ithaca, New York, USA</publisher-loc>: <publisher-name>ArXiv</publisher-name>).</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Multiple disease detection method for greenhouse-cultivated strawberry based on multiscale feature fusion faster r-cnn</article-title>. <source>Comput. Electron. Agric.</source> <volume>199</volume>, <elocation-id>107176</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.107176</pub-id>
</citation></ref>
</ref-list>
</back>
</article>