<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2024.1499278</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>TomatoGuard-YOLO: a novel efficient tomato disease detection method</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Xuewei</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/694621"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Jun</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/694621"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>Shandong Provincial University Laboratory for Protected Horticulture, Weifang University of Science and Technology</institution>, <addr-line>Weifang</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Dun Wang, Northwest A&amp;F University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Lingxian Zhang, China Agricultural University, China</p>
<p>Meichen Feng, Shanxi Agricultural University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Jun Liu, <email xlink:href="mailto:liu_jun860116@wfust.edu.cn">liu_jun860116@wfust.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>31</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1499278</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>09</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>12</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wang and Liu</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang and Liu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Tomatoes are highly susceptible to numerous diseases that significantly reduce their yield and quality, posing critical challenges to global food security and sustainable agricultural practices. To address the shortcomings of existing detection methods in accuracy, computational efficiency, and scalability, this study propose TomatoGuard-YOLO, an advanced, lightweight, and highly efficient detection framework based on an improved YOLOv10 architecture. The framework introduces two key innovations: the Multi-Path Inverted Residual Unit (MPIRU), which enhances multi-scale feature extraction and fusion, and the Dynamic Focusing Attention Framework (DFAF), which adaptively focuses on disease-relevant regions, substantially improving detection robustness. Additionally, the incorporation of the Focal-EIoU loss function refines bounding box matching accuracy and mitigates class imbalance. Experimental evaluations on a dedicated tomato disease detection dataset demonstrate that TomatoGuard-YOLO achieves an outstanding mAP50 of 94.23%, an inference speed of 129.64 FPS, and an ultra-compact model size of just 2.65 MB. These results establish TomatoGuard-YOLO as a transformative solution for intelligent plant disease management systems, offering unprecedented advancements in detection accuracy, speed, and model efficiency.</p>
</abstract>
<kwd-group>
<kwd>tomato disease detection</kwd>
<kwd>YOLOv10</kwd>
<kwd>multi-path inverted residual unit</kwd>
<kwd>dynamic focusing attention framework</kwd>
<kwd>focal-EIoU loss function</kwd>
</kwd-group>
<counts>
<fig-count count="12"/>
<table-count count="11"/>
<equation-count count="13"/>
<ref-count count="50"/>
<page-count count="19"/>
<word-count count="9410"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>With the advent of artificial intelligence (AI) and the global shift towards smart agriculture, precision farming techniques have emerged as essential tools for addressing crop disease challenges in modern agricultural production (<xref ref-type="bibr" rid="B23">Li et&#xa0;al., 2021</xref>). AI has significantly transformed the field of plant disease detection, enabling more accurate and timely diagnosis while becoming an increasingly integral part of agricultural practices, particularly in identifying crop diseases (<xref ref-type="bibr" rid="B43">Wang et&#xa0;al., 2020</xref>). Among the most widely cultivated crops globally, tomatoes are highly susceptible to various diseases such as late blight, leaf mold, and bacterial speck. These diseases can cause extensive damage, leading to severe plant wilting and significant reductions in yield. The economic impact of such outbreaks is substantial, underscoring the critical importance of timely detection and control measures to ensure food security and sustain agricultural productivity.</p>
<p>Traditional methods for disease detection often rely on visual diagnosis by experienced agronomists. While effective in certain contexts, these approaches are inherently subjective and prone to inconsistencies (<xref ref-type="bibr" rid="B20">Kaniyassery et&#xa0;al., 2024</xref>). Moreover, as agricultural production scales up, manual monitoring of crops becomes increasingly inefficient and imprecise, rendering it inadequate to meet the demands of large-scale farming operations (<xref ref-type="bibr" rid="B38">Singh et&#xa0;al., 2020</xref>).</p>
<p>To address these limitations, AI-based image recognition technologies have been increasingly utilized in crop disease identification. In particular, deep learning-based object detection algorithms have demonstrated significant potential in automating disease detection in both drone-assisted and greenhouse environments. This has driven research efforts toward developing efficient, accurate, and scalable automated disease detection systems that can enhance agricultural productivity and resilience against crop diseases.</p>
<p>Object detection, a fundamental research area within computer vision, primarily comprises traditional methods and deep learning-based approaches. Traditional methods, such as SIFT and HOG features paired with SVM for object detection (<xref ref-type="bibr" rid="B29">Lowe, 2004</xref>), continue to grapple with limitations in detection accuracy and generalization when confronted with complex backgrounds and varied disease symptoms (<xref ref-type="bibr" rid="B47">Zhang et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B26">Liu et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B50">Zhu et&#xa0;al., 2019</xref>).</p>
<p>The advancement of deep learning technology has significantly propelled the development of CNN-based object detection techniques for tomato disease recognition, encompassing Faster R-CNN (<xref ref-type="bibr" rid="B14">Girshick, 2015</xref>), SSD (<xref ref-type="bibr" rid="B27">Liu et&#xa0;al., 2016</xref>), and YOLO (<xref ref-type="bibr" rid="B33">Redmon et&#xa0;al., 2016</xref>) (<xref ref-type="bibr" rid="B11">Fuentes et&#xa0;al., 2017</xref>). Among these, the YOLO series of algorithms has garnered substantial attention owing to its speed and accuracy. With the introduction of YOLOv4 (<xref ref-type="bibr" rid="B4">Bochkovskiy et&#xa0;al., 2020</xref>), YOLOv5 (<xref ref-type="bibr" rid="B16">Jocher et&#xa0;al., 2022</xref>), YOLOX (<xref ref-type="bibr" rid="B13">Ge et&#xa0;al., 2021</xref>), YOLOv6 (<xref ref-type="bibr" rid="B22">Li et&#xa0;al., 2022</xref>), YOLOv7 (<xref ref-type="bibr" rid="B42">Wang et&#xa0;al., 2023</xref>), and other epoches, performance levels have consistently improved. The continuous evolution of YOLO algorithms demonstrates significant improvements in object detection capabilities through systematic architecture optimization and technical innovations. The core architectural enhancements include implementing more efficient backbone networks for feature extraction, optimizing neck structures for better feature fusion, introducing advanced head designs for more accurate predictions, and incorporating attention mechanisms for improved feature focus. These fundamental improvements are complemented by technical innovations such as enhanced loss functions for better convergence, adaptive feature aggregation for multi-scale detection, improved anchor-free detection mechanisms, and advanced data augmentation strategies.</p>
<p>The latest YOLOv10 algorithm has achieved innovations in multiple aspects, significantly enhancing detection accuracy and speed (<xref ref-type="bibr" rid="B41">Wang et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B1">Alif and Hussain, 2024</xref>). Specifically, the BGF-YOLOv10 algorithm, designed for small object detection, achieved a remarkable mAP of 42.0% on the VisDroneDET2019 dataset, demonstrating significant improvement over earlier versions (<xref ref-type="bibr" rid="B31">Mei and Zhu, 2024</xref>). Additionally, the BLP-YOLOv10 model, optimized for safety helmet detection in low-light environments, achieved an impressive mAP of 98.1% (<xref ref-type="bibr" rid="B10">Du et&#xa0;al., 2024</xref>). This model excels in feature extraction and image processing by adjusting backbone channel parameters, incorporating sparse attention mechanisms, and integrating low-frequency enhancement filters.</p>
<p>Nevertheless, YOLOv10 still faces challenges in small object detection, complex background interference, and multi-scale target handling (<xref ref-type="bibr" rid="B19">Kang et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B21">Lee et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B36">Sangaiah et&#xa0;al., 2024</xref>). Compared to other versions, the decision to improve YOLOv10 was based on several key factors. First, YOLOv10 introduces a lightweight architecture and multi-path convolution modules, significantly enhancing adaptability in complex environments while maintaining precision and optimizing computational efficiency, making it suitable for real-time, resource-constrained scenarios. Second, its modular design provides high flexibility for integrating novel mechanisms such as adaptive attention and inverted residual units, making it well-suited for large-scale disease detection tasks. Lastly, YOLOv10 has demonstrated excellent performance in practical applications. For example, BGF-YOLOv10 has improved crop disease detection in agricultural scenarios, while BLP-YOLOv10 has shown exceptional robustness in industrial settings under challenging lighting conditions (<xref ref-type="bibr" rid="B31">Mei and Zhu, 2024</xref>; <xref ref-type="bibr" rid="B10">Du et&#xa0;al., 2024</xref>). These comprehensive improvements and remaining challenges make YOLOv10 an ideal candidate for continued development and optimization, particularly in addressing specific application requirements while maintaining general-purpose detection capabilities. The balance between addressing current limitations and leveraging existing strengths positions YOLOv10 as a promising framework for future advancements in object detection technology.</p>
<p>To significantly enhance the accuracy and efficiency of tomato disease object detection, this study proposes an improved algorithm based on YOLOv10. The research framework comprises three key stages. First, a self-built tomato disease dataset is constructed to provide high-quality input data, including standardized annotation and preprocessing. Second, during model construction, the proposed Multi-Path Inverted Residual Unit (MPIRU) and Dynamic Focusing Attention Framework (DFAF) enhance feature extraction and small target detection capabilities. Additionally, the optimization of the loss function and class balance strategies improves overall detection performance. Finally, the effectiveness of the enhanced model is validated through comparative experiments with existing models. This study aims to provide robust technical support for early detection and precise control of tomato diseases, contributing to the development of intelligent agriculture and ensuring food safety and agricultural production efficiency.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Literature review</title>
<sec id="s2_1">
<label>2.1</label>
<title>Agricultural disease detection: from traditional methods to deep learning</title>
<p>In recent years, artificial intelligence and deep learning technologies have been gradually applied to various agricultural domains, such as crop cultivation, harvesting, and disease detection (<xref ref-type="bibr" rid="B18">Kamilaris and Prenafeta-Bold&#xfa;, 2018</xref>). Farmers have begun using smartphones to detect crop diseases and pests (<xref ref-type="bibr" rid="B28">Liu et&#xa0;al., 2020</xref>). However, this field still faces numerous challenges and development opportunities.</p>
<p>Traditional agricultural disease detection and identification methods primarily rely on manual feature extraction. While these methods have achieved certain research results (<xref ref-type="bibr" rid="B7">Camargo and Smith, 2009</xref>; <xref ref-type="bibr" rid="B37">Shekhawat and Sinha, 2020</xref>), they have apparent limitations. These conventional approaches necessitate substantial professional expertise and a deep reservoir of knowledge, are inherently subjective, and often neglect valuable attributes that are challenging to identify with unaided human perception. Furthermore, when confronted with voluminous data in authentic natural settings, the precision of traditional methods frequently deteriorates significantly (<xref ref-type="bibr" rid="B6">Buja et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B44">Wiesner-Hanks et&#xa0;al., 2019</xref>).</p>
<p>In contrast, deep learning technologies, with their potent feature representation capacities, can autonomously extract features from voluminous multi-type disease data, thereby substantially enhancing detection performance (<xref ref-type="bibr" rid="B34">Saleem et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B3">Bhattacharya et&#xa0;al., 2022</xref>). AI has made disease detection more automated, efficient, precise, and reliable, primarily manifested in the following aspects:</p>
<list list-type="order">
<list-item>
<p>Automatic feature learning: Capable of automatically learning and recognizing disease features from millions of images, greatly enhancing detection accuracy.</p>
</list-item>
<list-item>
<p>No manual intervention: Does not require manual feature extraction or threshold setting, can automatically adapt to various environments and conditions.</p>
</list-item>
<list-item>
<p>Efficient processing: Capable of rapidly identifying a substantial quantity of diseases, thereby enhancing detection efficiency and accuracy.</p>
</list-item>
</list>
<p>The efficacy of deep learning models in plant disease detection is intrinsically tied to the quality of the training data. Currently, the creation of datasets for agricultural disease detection models can be broadly classified into two primary categories: natural environment photography (with background) and ideal environment (without background), as depicted in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>. Images captured in natural environments present complex backgrounds, which can enhance the robustness and generalization capabilities of trained models. Conversely, images acquired in ideal environments lack background distractions, but the resulting models often struggle&#xa0;to achieve satisfactory detection performance in real-world scenarios.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Dataset examples. <bold>(A)</bold> natural environment photography (with background) <bold>(B)</bold> ideal environment (without background).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g001.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Agricultural disease detection in ideal environments: achievements and limitations</title>
<p>Given the intricate relationship between agricultural diseases and factors like cultivation practices, management strategies, and climate fluctuations, prevailing open-source tomato disease datasets largely depend on laboratory samples, exemplified by AI Challenger 2018, Kaggle, and PlantVillage. Recognizing the considerable time and effort necessary to accumulate a substantial quantity of natural environment samples, numerous agricultural disease detection model studies predominantly leverage open-source ideal environment samples for training purposes. Although models developed based on ideal environment samples have achieved notable outcomes in laboratory settings, many have not undergone validation in natural environments. Existing research suggests that these models are primarily well-suited for scenarios where disease-affected areas constitute a significant portion of the image but may encounter difficulties in handling complex backgrounds, lighting variations, changes in shooting angles, and diverse lesion sizes within natural scenes.</p>
<p>For instance, <xref ref-type="bibr" rid="B5">Bora et&#xa0;al. (2023)</xref> developed a system capable of detecting diseases on tomato leaves, stems, fruits, and roots with remarkable accuracy rates of 99.84%, 95.2%, 96.8%, and 93.6%, respectively. <xref ref-type="bibr" rid="B48">Zhang et&#xa0;al. (2023)</xref> introduced the M-AORANet model, which demonstrated exceptional recognition accuracy of 96.47% on a dataset comprising 3,123 tomato leaf images. <xref ref-type="bibr" rid="B40">Sunil et&#xa0;al. (2023)</xref> employed a Multi-level Feature Fusion Network (MFFN) to achieve an impressive external test accuracy of 99.83% on publicly available tomato disease datasets. Although these models exhibited exceptional performance in controlled environments, their primary limitation lies in their inability to pinpoint the exact location of lesions within images, hindering their direct application in real-world agricultural settings.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Agricultural disease detection in natural environments: progress and challenges</title>
<p>Although models trained on natural environment data more accurately reflect real-world conditions, the majority of research continues to concentrate on model development, refinement, and structural analysis using personal computers. This neglects the critical need for lightweight and highly accurate solutions in practical applications.</p>
<p>Several prior studies have explored the use of deep learning for plant disease detection in various agricultural settings. <xref ref-type="bibr" rid="B24">Li et&#xa0;al. (2019)</xref> developed a mobile application for early detection of tomato late blight, demonstrating the potential for smartphone-based disease diagnosis. <xref ref-type="bibr" rid="B39">Sun et&#xa0;al. (2021)</xref> proposed the MEAN-SSD model, achieving an accuracy of 83.12% for apple leaf disease detection while maintaining real-time processing speeds (12.53 frames per second). <xref ref-type="bibr" rid="B46">Zhang K. et&#xa0;al. (2021)</xref> incorporated skip connections within the Faster R-CNN architecture, achieving an accuracy of 83.34% on a custom dataset of soybean disease images. Similarly, <xref ref-type="bibr" rid="B8">Chen et&#xa0;al. (2021)</xref> constructed a model for cucumber leaf disease detection with an accuracy of 85.52%. Finally, <xref ref-type="bibr" rid="B9">Dananjayan et&#xa0;al. (2022)</xref> demonstrated the effectiveness of YOLOv4 for rapid and accurate detection of citrus leaf diseases. <xref ref-type="bibr" rid="B32">Qi et&#xa0;al. (2022)</xref> introduced an enhanced SE-YOLOv5s network model that achieved an impressive 91.07% accuracy on the tomato disease test set. While these studies demonstrated real-time disease recognition capabilities, the models developed for individual agricultural diseases face challenges in widespread deployment due to the fluctuating nature of disease occurrences.</p>
<p>Machine vision detection of tomato diseases faces significant obstacles in actual planting environments, including complex growing conditions, multiple disease types, and subtle symptom variations (<xref ref-type="bibr" rid="B2">Barbedo, 2018</xref>; <xref ref-type="bibr" rid="B17">Kadry, 2021</xref>; <xref ref-type="bibr" rid="B12">Fuentes et&#xa0;al., 2021</xref>), which impose exceptionally high demands on the multi-feature and cross-scale extraction capabilities of detection algorithms. While the YOLO series models have garnered widespread adoption for their swift and precise detection capabilities, there remains potential for enhancement in feature extraction and detection accuracy within complex environments.</p>
<p>Considering the current research status and challenges, this study proposes a tomato disease detection method based on a refined YOLOv10 architecture. By meticulously analyzing tomato disease types and image characteristics, the algorithm is iteratively enhanced and experimentally validated, aiming to fulfill the accuracy and speed requirements of intelligent tomato disease detection and reduce manual diagnosis costs. The innovations of this study are primarily as follows:</p>
<list list-type="order">
<list-item>
<p>The introduction of the Multi-Path Inverted Residual Unit (MPIRU) significantly enhances the model&#x2019;s ability to fuse multi-scale features through parallel processing across multiple paths, effectively reducing the number of model parameters.</p>
</list-item>
<list-item>
<p>Integration of a Dynamic Focusing Attention Framework (DFAF) into the C2f module, improving the focus on important target areas and localization accuracy.</p>
</list-item>
<list-item>
<p>By incorporating Focal-EIoU as a refined loss function, we significantly enhanced the model&#x2019;s ability to accurately match objects while effectively mitigating the challenges posed by imbalanced datasets.</p>
</list-item>
<list-item>
<p>The improved YOLOv10 performs excellently in detection accuracy, parameter optimization, and complexity control, making it suitable not only for tomato disease detection but&#xa0;also for other crop disease detection tasks in complex backgrounds.</p>
</list-item>
</list>
</sec>
</sec>
<sec id="s3" sec-type="materials|methods">
<label>3</label>
<title>Materials and methods</title>
<sec id="s3_1">
<label>3.1</label>
<title>Data collection and dataset preparation</title>
<p>In this study, we utilized a custom-built tomato disease dataset aimed at capturing disease features in real agricultural environments, reflecting the imbalanced nature of disease occurrence in actual scenarios. The data collection process was rigorously controlled to ensure quality and representativeness. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> summarizes the key parameters of data collection.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Overview of data collection parameters.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Parameter</th>
<th valign="middle" align="center">Description</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Collection equipment</td>
<td valign="middle" align="center">Agricultural IoT monitoring equipment (HS-CQAI-1080)</td>
</tr>
<tr>
<td valign="middle" align="center">Location</td>
<td valign="middle" align="center">Tomato production base, Shouguang City, Shandong Province, China</td>
</tr>
<tr>
<td valign="middle" align="center">Precise coordinates</td>
<td valign="middle" align="center">Longitude: 118.782956&#xb0;, Latitude: 36.930686&#xb0;</td>
</tr>
<tr>
<td valign="middle" align="center">Image resolution</td>
<td valign="middle" align="center">3648 &#xd7; 2056 pixels</td>
</tr>
<tr>
<td valign="middle" align="center">Daily collection times</td>
<td valign="middle" align="center">08:30&#x2013;11:30 and 14:30&#x2013;17:30</td>
</tr>
<tr>
<td valign="middle" align="center">Distance between device and lesions</td>
<td valign="middle" align="center">0.2~0.5 meters</td>
</tr>
<tr>
<td valign="middle" align="center">Environmental conditions</td>
<td valign="middle" align="center">Sunny and cloudy days; various infected regions and conditions</td>
</tr>
<tr>
<td valign="middle" align="center">Total number of images</td>
<td valign="middle" align="center">10,537</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The data acquisition process prioritized capturing a broad spectrum of environmental conditions, encompassing diverse lighting scenarios, angles of view, and background elements, such as leaves, weeds, and soil. This diversity is crucial for improving the model&#x2019;s generalization ability and applicability in real-world conditions. Each image was accompanied by rich metadata, including environmental temperature, precise location, and timestamp, providing valuable context for subsequent analysis and model training. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> shows representative samples of the various tomato diseases in the dataset, visually illustrating the characteristics and complexity of different diseases.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Shows five samples of the various tomato diseases in the dataset: <bold>(A)</bold> Late blight <bold>(B)</bold> Early blight <bold>(C)</bold> Gray mold <bold>(D)</bold> Leaf mold <bold>(E)</bold> Health.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g002.tif"/>
</fig>
<p>Our dataset reflects the natural distribution of disease occurrence in actual agricultural environments, thus exhibiting significant class imbalance. <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> details the sample count for each category and its proportion in the overall dataset.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Sample count for each category of our dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Category</th>
<th valign="middle" align="center">Sample count</th>
<th valign="middle" align="center">Proportion</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">Healthy</td>
<td valign="middle" align="center">3526</td>
<td valign="middle" align="center">33.46%</td>
</tr>
<tr>
<td valign="bottom" align="center">Late Blight</td>
<td valign="middle" align="center">2745</td>
<td valign="middle" align="center">26.05%</td>
</tr>
<tr>
<td valign="bottom" align="center">Early Blight</td>
<td valign="middle" align="center">2103</td>
<td valign="middle" align="center">19.96%</td>
</tr>
<tr>
<td valign="bottom" align="center">Gray Mold</td>
<td valign="middle" align="center">1358</td>
<td valign="middle" align="center">12.89%</td>
</tr>
<tr>
<td valign="bottom" align="center">Leaf Mold</td>
<td valign="middle" align="center">805</td>
<td valign="middle" align="center">7.64%</td>
</tr>
<tr>
<td valign="top" align="center">Total</td>
<td valign="middle" align="center">10,537</td>
<td valign="middle" align="center">100%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>This imbalanced distribution reflects the relative frequency of various diseases in actual agricultural production, providing us with a realistic challenge scenario. In particular, the sample sizes for leaf mold and gray mold are notably smaller than other categories, highlighting the importance and difficulty of identifying rare diseases in practical applications.</p>
<p>To address the dataset&#x2019;s class imbalance, we employed a stratified random sampling technique. This method ensured that the distribution of each category in the training, validation, and testing sets accurately mirrored the original dataset. A detailed breakdown of the dataset division is provided in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Dataset division.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Dataset</th>
<th valign="middle" align="center">Sample size</th>
<th valign="middle" align="center">Proportion</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Training set</td>
<td valign="middle" align="center">8430</td>
<td valign="middle" align="center">80%</td>
</tr>
<tr>
<td valign="top" align="center">Validation set</td>
<td valign="middle" align="center">1054</td>
<td valign="middle" align="center">10%</td>
</tr>
<tr>
<td valign="top" align="center">Test set</td>
<td valign="middle" align="center">1053</td>
<td valign="middle" align="center">10%</td>
</tr>
<tr>
<td valign="top" align="center">Total</td>
<td valign="middle" align="center">10,537</td>
<td valign="middle" align="center">100%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>This division method ensures that each subset contains balanced representations of all categories while preserving the imbalanced characteristics of the original dataset, contributing to stable performance and reliable evaluation of the model in real-world scenarios.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Data annotation</title>
<p>Accurate data annotation is indispensable for guaranteeing the efficacy of model training. This research employed the widely utilized open-source annotation tool Labellmg to prepare data for object detection purposes. The annotation process is visually depicted in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Data Annotation Process. Using LabelImg tool for disease area annotation. Example of generated VOC format XML file.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g003.tif"/>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Data augmentation</title>
<p>Given the dataset&#x2019;s uneven distribution, we implemented focused data augmentation techniques designed to address class imbalance concerns and bolster the model&#x2019;s capacity to identify uncommon classes.</p>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Offline data augmentation</title>
<p>To improve model generalization and mitigate overfitting, this study performed data augmentation on the training set. When selecting data augmentation methods, we paid special attention to preserving key disease features while avoiding unnecessary distortions. We adopted more aggressive offline data augmentation strategies for categories with fewer samples (such as leaf mold and gray mold). <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> outlines the offline augmentation methods and their parameters applied to different categories.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Offline data augmentation methods (for different categories).</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Category</th>
<th valign="middle" align="center">Augmentation Methods</th>
<th valign="middle" align="center">Parameters</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Leaf mold</td>
<td valign="middle" align="center">Horizontal flip, Vertical flip, Small angle rotation, Brightness adjustment</td>
<td valign="middle" align="center">Rotation range: &#xb1; 20&#xb0;, Brightness range: &#xb1; 15%</td>
</tr>
<tr>
<td valign="middle" align="center">Gray mold</td>
<td valign="middle" align="center">Horizontal flip, Vertical flip, Small angle rotation</td>
<td valign="middle" align="center">Rotation range: &#xb1; 15&#xb0;, Brightness range: &#xb1; 10%</td>
</tr>
<tr>
<td valign="middle" align="center">Early blight, Late blight</td>
<td valign="middle" align="center">Horizontal flip, Vertical flip</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="center">Healthy samples</td>
<td valign="middle" align="center">Horizontal flip</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> demonstrates the effects of four image processing techniques. Through this strategy, we significantly increased the sample size of rare categories while maintaining the overall diversity of the dataset. This approach helps balance the performance across categories, especially improving the model&#x2019;s ability to recognize relatively rare diseases such as leaf mold and gray mold.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Offline data augmentation examples. <bold>(A)</bold> Original image; <bold>(B)</bold> Horizontal flip; <bold>(C)</bold> Vertical flip; <bold>(D)</bold> Small angle rotation; <bold>(E)</bold> Brightness adjustment.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g004.tif"/>
</fig>
<p>Through the offline data augmentation strategy, we significantly expanded the total number of samples. The original dataset contained 10,537 images, which increased to 31,053 after augmentation, approximately 2.95 times the original size. Notably, for rare categories such as leaf mold and gray mold, the sample counts increased from 805 and 1358 to 4025 and 5432, respectively. This category-specific augmentation approach effectively mitigated the class imbalance issue, improved the model&#x2019;s ability to recognize rare categories, and preserved critical disease features, thereby providing a more robust foundation for the model&#x2019;s generalization performance.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Real-time data augmentation</title>
<p>Given the dataset&#x2019;s uneven distribution and the intricate agricultural setting, we implemented a variety of real-time data augmentation techniques within YOLOv10 with the objective of enhancing the model&#x2019;s capacity for generalization and its aptitude for identifying uncommon categories (<xref ref-type="table" rid="T5">
<bold>Table 5</bold>
</xref>).</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Real-time data augmentation methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Method</th>
<th valign="middle" align="center">Parameters</th>
<th valign="middle" align="center">Purpose</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Mosaic</td>
<td valign="middle" align="center">Probability = 1.0, Number of images = 4</td>
<td valign="middle" align="center">Increase contextual information, improve small object detection performance</td>
</tr>
<tr>
<td valign="middle" align="center">Random Affine</td>
<td valign="middle" align="center">Rotation = &#xb1; 10&#xb0;, Scale = 0.8~1.2</td>
<td valign="middle" align="center">Simulate different shooting angles and distances</td>
</tr>
<tr>
<td valign="middle" align="center">MixUp</td>
<td valign="middle" align="center">Probability = 0.15</td>
<td valign="middle" align="center">Increase sample diversity, improve model generalization ability</td>
</tr>
<tr>
<td valign="middle" align="center">Random HSV</td>
<td valign="middle" align="center">Hue = &#xb1; 10, Saturation = 0.5, Value = 0.5</td>
<td valign="middle" align="center">Simulate different lighting and weather conditions</td>
</tr>
<tr>
<td valign="middle" align="center">Cutout</td>
<td valign="middle" align="center">Probability = 0.3</td>
<td valign="middle" align="center">Improve model robustness to partial occlusions</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>This all-encompassing real-time data augmentation approach not only substantially expands the diversity of the training dataset but also enhances the model&#x2019;s capacity to adapt to a wide range of intricate scenarios. Especially for disease categories with fewer samples (such as leaf mold and gray mold), these enhancement methods help the model learn more diverse feature representations from limited samples. Through this approach, we expect the model to better handle the complex and variable real-world scenarios in agricultural production, improving detection accuracy for various diseases, particularly under challenging conditions such as insufficient lighting, partial occlusion, or unfavorable shooting angles.</p>
</sec>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>The improved tomato disease detection model based on YOLOv10</title>
<p>Although the C2f module in YOLOv10 enhances feature extraction capabilities through the Bottleneck structure, its equal treatment of all channels and positional information introduces a significant amount of irrelevant interference. This results in suboptimal performance when handling multi-scale small targets and complex backgrounds in tomato disease images. Additionally, in the YOLOv10 model, the backbone network extracts deep features through multiple down-sampling convolution layers.</p>
<p>Although this multi-stage downsampling enhances the model&#x2019;s capacity to handle large targets and intricate backgrounds, it also leads to a significant reduction in small target features. To mitigate the loss of these crucial details during detection, the neck network utilizes multiple upsampling operations to restore feature map resolution and integrate features from various levels, thereby improving the model&#x2019;s ability to detect targets of varying sizes. However, this alternating process of downsampling and upsampling also results in excessive layer stacking within the backbone and neck networks, increasing the model&#x2019;s parameter count and computational complexity, making it challenging to meet real-time detection requirements in practical applications. To overcome these aforementioned challenges, given the nature of multi-scale target detection of tomato diseases in intricate environments, this study proposes a streamlined target detection algorithm, TomatoGuard-YOLO, as illustrated in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>The architecture of the proposed TomatoGuard-YOLO model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g005.tif"/>
</fig>
<p>This model is an enhancement of version n of YOLOv10 (referred to as YOLOv10 unless otherwise specified), featuring lightweight designs for both the backbone and neck networks. Initially, a novel feature extraction and fusion module, termed the Multi-Path Inverted Residual Unit (MPIRU), is devised. Subsequently, the proposed Dynamic Focusing Attention Framework (DFAF) is integrated into the C2f module, resulting in the C2f-DFAF module. Subsequently, all C2f modules within the foundational YOLOv10 backbone and neck networks are supplanted with MCIR and C2f-DFAF. Furthermore, the backbone network&#x2019;s downsampling operations have been curtailed to retain a greater quantity of feature information. Concurrently, the upsampling and feature concatenation procedures within the neck network have been streamlined to diminish the number of layers and complexity, thereby further reducing computational expenses. In conclusion, the loss function has been meticulously refined to Focal-EIoU, effectively rectifying the deficiencies of the original loss function and bolstering the model&#x2019;s capacity to concentrate on a variety of samples.</p>
<p>The subsequent sections elucidate each improved module, aiming to elevate the model&#x2019;s performance in detecting tomato diseases in intricate agricultural environments, particularly in handling small targets, complex backgrounds, and class imbalance challenges.</p>
<sec id="s3_4_1">
<label>3.4.1</label>
<title>Multi-path inverted residual unit</title>
<p>In complex agricultural environments, tomato diseases often exhibit multi-scale and multi-form characteristics. To bolster the model&#x2019;s capacity to extract these intricate features, we introduce the Multi-Path Inverted Residual Unit (MPIRU). The design of MPIRU incorporates the inverted residual structure from MobileNetV2 (<xref ref-type="bibr" rid="B35">Sandler et&#xa0;al., 2018</xref>) and the channel separation concept from ShuffleNetV2 (<xref ref-type="bibr" rid="B30">Ma et&#xa0;al., 2018</xref>), aiming to enhance feature extraction diversity while preserving computational efficiency. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> illustrates the detailed structure of MPIRU.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Structure of the Multi-Path Inverted Residual Unit (MPIRU).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g006.tif"/>
</fig>
<p>As shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>, MPIRU first evenly divides the input feature map (X) into (n) branches {X<sub>1</sub>, X<sub>2</sub>,&#x2026;, X<sub>n</sub>}, with each branch independently processing a portion of the channels. Each branch adopts an &#x201c;expand-convolve-squeeze&#x201d; inverted residual structure, which can be expressed as:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>E<sub>i</sub>
</italic>, <italic>G<sub>i</sub>
</italic>, and <italic>H<sub>i</sub>
</italic> represent the expansion (1x1 convolution), depthwise separable convolution (3x3), and compression (1x1 convolution) operations, respectively. The output (Y) of MPIRU can be expressed as:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where Concat denotes the concatenation operation along the channel dimension.</p>
<p>To facilitate information exchange between different branches, we apply a channel shuffle operation after concatenation. This design not only enhances the diversity of feature extraction but also maintains relative computational stability.</p>
</sec>
<sec id="s3_4_2">
<label>3.4.2</label>
<title>C2f-DFAF module</title>
<p>Accurately locating and identifying diseased areas is essential in disease detection tasks. Attention mechanisms have proven highly effective in enhancing the architecture of deep neural networks, achieving notable success in various applications. However, their integration into lightweight networks has significantly lagged behind their implementation in larger models. This disparity arises primarily because most mobile and edge devices have limited computational resources, making it challenging to accommodate the high overhead associated with traditional attention mechanisms. To address this limitation, we propose the Dynamic Focusing Attention Framework (DFAF), which is seamlessly integrated with the C2f module to form the novel C2f-DFAF module, delivering efficient and effective attention capabilities suitable for lightweight networks.</p>
<p>This enhancement is inspired by SENet (<xref ref-type="bibr" rid="B15">Hu et&#xa0;al., 2018</xref>) and CBAM (<xref ref-type="bibr" rid="B45">Woo et&#xa0;al., 2018</xref>). SENet employs 2D global pooling to calculate channel attention and improves model performance with a relatively minimal computational overhead. Nevertheless, SENet solely considers inter-channel information and neglects positional information, which is essential for capturing object structures in visual tasks. To address this limitation, CBAM endeavors to compute positional information by reducing the channel dimension of the input tensor and subsequently utilizing convolution to calculate spatial attention. However, convolution solely captures local area information and cannot model long-distance dependencies, nor can it capture spatial information at varying scales to enrich the feature space. Although CBAM uses two fully connected layers and a nonlinear Sigmoid function to generate channel weights, capturing nonlinear cross-channel interaction information and controlling model complexity through dimensionality reduction, the parameter count is still positively correlated with the square of the input feature map channels. Further research shows that dimensionality reduction negatively impacts channel attention prediction, with low efficiency in capturing dependencies among all channels.</p>
<p>To address the critical challenges in object detection, we introduce the C2f-DFAF module, centered around a lightweight adaptive attention unit (DFAF) capable of dynamically learning and adjusting the significance of features across channel and spatial dimensions. DFAF effectively pinpoints tomato disease features, suppresses irrelevant features, and significantly reduces parameters and computational overhead. The module comprises a feature input layer, channel attention module layer, spatial attention module layer, and feature output layer. <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> illustrates the detailed structure of the DFAF module.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Structure of DFAF attention mechanism.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g007.tif"/>
</fig>
<p>The DFAF (Dynamic Feature Adaptive Fusion) module consists of two primary components: a channel attention module and a spatial attention module. The input features undergo parallel processing through these modules, with their outputs being adaptively fused through multiplication operations. A skip connection preserves the original feature information, ensuring robust feature representation. This architecture enables dynamic feature weighting while maintaining computational efficiency. Specifically, the feature input layer is responsible for receiving and pre-processing raw input features, preparing them for subsequent attention mechanism modules. The channel attention module layer learns weights for each channel, adaptively enhancing important feature channels while suppressing secondary channels, thereby modeling the global importance of features. The spatial attention module layer focuses on capturing spatial dependencies within feature maps by generating attention weight maps, precisely localizing tomato disease regions. Finally, the feature output layer integrates the outputs of the aforementioned attention mechanisms, generating more focused and discriminative feature representations.</p>
<p>The mathematical expression of the DFAF module is as follows:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#x2297;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mo>&#x2297;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In the aforementioned formular, <italic>X</italic> represents the input feature, <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the weight parameters for channel and spatial attention, respectively, &#x3c3; is the sigmoid activation function, and <inline-formula>
<mml:math display="inline" id="im3">
<mml:mo>&#x2297;</mml:mo>
</mml:math>
</inline-formula> denotes the convolution operation. This design allows the model to adaptively balance the significance of channel and spatial attention, thereby better accommodating different types of disease features.</p>
<p>Simultaneously, the C2f-DFAF module introduces a residual learning mechanism. As shown in <xref ref-type="disp-formula" rid="eq4">Equation 4</xref>, the output of the attention mechanism is added to the original features rather than simply multiplied:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>Y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo>*</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>This design helps mitigate the vanishing gradient problem while preserving the original feature information, which is particularly important for maintaining the subtle features of diseases. Consequently, the DFAF module utilizes the channel attention module to generate channel-level attention maps, thereby enhancing the response to small targets. The spatial attention module generates spatial-level attention maps through convolution operations, enabling the model to accurately pinpoint regions of interest in intricate backgrounds. By integrating these two attention mechanisms, the model can adjust and weight channel and spatial attention, allowing it to concentrate on significant regions in complex backgrounds, strengthening small target detection capabilities, and improving recognition accuracy. Additionally, the use of global pooling and convolution operations enables efficient parallel computation without adding too many extra parameters, allowing the C2f-DFAF module to achieve excellent detection performance with high efficiency and low parameters in tomato disease detection.</p>
</sec>
<sec id="s3_4_3">
<label>3.4.3</label>
<title>Focal-EIoU loss function</title>
<p>Tomato disease samples in images often exhibit substantial class imbalance, with significant variations in shape and size. To address these challenges, we propose the Focal-EIoU loss function, which integrates the sample balancing capability of Focal Loss (<xref ref-type="bibr" rid="B25">Lin et&#xa0;al., 2017</xref>) and the precise bounding box regression capability of EIoU (Efficient IoU) (<xref ref-type="bibr" rid="B49">Zhang Z. et&#xa0;al., 2021</xref>).</p>
<p>Focal Loss is particularly effective in addressing class imbalance issues by dynamically reducing the loss weight of easily classified samples, thereby increasing the model&#x2019;s emphasis on difficult-to-classify samples. The definition of Focal Loss is as follows:</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:msup>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>&#x3bd;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>4</mml:mn>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtext>arctan</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>arctan</mml:mtext>
<mml:mfrac>
<mml:mi>&#x3c9;</mml:mi>
<mml:mi>h</mml:mi>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The model&#x2019;s predicted probability for the accurate classification is denoted by <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. To address the issue of imbalanced classes, we employ a weighting factor represented by <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Additionally, the parameter &#x3b3; serves to regulate the influence of easily classified samples. The dimensions of the ground truth and predicted bounding boxes are expressed as (<inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c9;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>) and (<inline-formula>
<mml:math display="inline" id="im8">
<mml:mi>&#x3c9;</mml:mi>
</mml:math>
</inline-formula>, <italic>h</italic>), respectively. The central coordinates of the predicted and ground truth boxes are given by <italic>b</italic> and <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The Euclidean distance between these center points is calculated as <italic>&#x3c1;</italic>. The diagonal length of the smallest bounding box encompassing both boxes is represented by <italic>c</italic>. The weighting function is defined as <italic>&#x3b1;</italic>, and the squared disparity in the diagonal angles of the ground truth and predicted boxes is denoted by <italic>v</italic>, as illustrated in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Loss function border diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g008.tif"/>
</fig>
<p>Focal Loss effectively prioritizes challenging samples with low intersection over union (IoU) by assigning them greater weights. This mechanism empowers the model to concentrate on more elusive classes, which is indispensable for the identification of uncommon diseases such as leaf mold and gray mold.</p>
<p>EIoU significantly elevates the accuracy of bounding box regression by refining the precision of the overlap between the predicted and ground truth boxes. EIoU comprehensively evaluates not only the area of overlap but also the relative positions, dimensions, and contours of the boxes. Its formal definition is as follows:</p>
<disp-formula id="eq6a">
<label>(6)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mi>&#x3bd;</mml:mi>
<mml:mi>c</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>EIoU provides a more comprehensive evaluation of bounding boxes than traditional IoU by penalizing deviations in position and aspect ratio, improving the localization accuracy for small targets and irregularly shaped diseases.</p>
<p>The final Focal-EIoU loss function integrates Focal Loss and EIoU, enabling the model to effectively mitigate class imbalance concerns and refine the accuracy of bounding box localization. Its mathematical formulation is as follows:</p>
<disp-formula id="eq7a">
<label>(7)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where &#x3bb; is the weighting factor for balancing classification and localization losses. After multiple experiments, the parameter &#x3b3;=0.9 was selected. The incorporation of Focal-EIoU loss into the TomatoGuard-YOLO model enables it to effectively prioritize less prevalent disease categories and accurately pinpoint the affected regions within intricate agricultural settings.</p>
</sec>
</sec>
</sec>
<sec id="s4" sec-type="results">
<label>4</label>
<title>Results</title>
<sec id="s4_1">
<label>4.1</label>
<title>Experimental environment</title>
<p>To ensure the reproducibility and reliability of the experimental results, this study meticulously records the experimental environment and key parameter settings. <xref ref-type="table" rid="T6">
<bold>Tables&#xa0;6</bold>
</xref>, <xref ref-type="table" rid="T7">
<bold>7</bold>
</xref> list the hardware and software configurations, as well as the core parameters for model training.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Experimental environment configuration.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Category</th>
<th valign="middle" align="center">Component</th>
<th valign="middle" align="center">Specification</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="4" align="center">Hardware</td>
<td valign="middle" align="center">Processor</td>
<td valign="middle" align="center">2 &#xd7; Intel Xeon Platinum 8280 (28 cores, 2.70 GHz)</td>
</tr>
<tr>
<td valign="middle" align="center">Memory</td>
<td valign="middle" align="center">768 GB DDR4-2933 ECC</td>
</tr>
<tr>
<td valign="middle" align="center">GPU</td>
<td valign="middle" align="center">4 &#xd7; NVIDIA A100 (80 GB HBM2e)</td>
</tr>
<tr>
<td valign="middle" align="center">Storage</td>
<td valign="middle" align="center">2 TB NVMe SSD + 20 TB HDD (RAID 5)</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="center">Software</td>
<td valign="middle" align="center">CUDA</td>
<td valign="middle" align="center">CUDA 11.4.2, cuDNN 8.2.4</td>
</tr>
<tr>
<td valign="middle" align="center">Python</td>
<td valign="middle" align="center">Python 3.9.7</td>
</tr>
<tr>
<td valign="middle" align="center">Framework</td>
<td valign="middle" align="center">PyTorch 1.10.1</td>
</tr>
<tr>
<td valign="middle" align="center">Libraries</td>
<td valign="middle" align="center">NumPy 1.21.4, OpenCV 4.5.4, Albumentations 1.1.0</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Model training parameters.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" align="center">Parameter</th>
<th valign="bottom" align="center">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">Batch Size</td>
<td valign="bottom" align="center">64</td>
</tr>
<tr>
<td valign="bottom" align="center">Initial Learning Rate</td>
<td valign="bottom" align="center">0.001</td>
</tr>
<tr>
<td valign="bottom" align="center">Weight Decay</td>
<td valign="bottom" align="center">0.05</td>
</tr>
<tr>
<td valign="bottom" align="center">Number of Epochs</td>
<td valign="bottom" align="center">300</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>This experiment employs cutting-edge hardware and software configurations, particularly the NVIDIA A100 GPU, whose powerful computing capabilities significantly enhance the efficiency of large-scale model training. The deep learning framework used is PyTorch, a popular open-source platform that seamlessly integrates with the hardware to optimize performance. Additionally, the selected versions of CUDA and cuDNN ensure full utilization of GPU acceleration and provide robust support for neural networks.</p>
<p>These parameters were chosen based on multiple experiments and optimizations. A batch size of 64 effectively utilizes the parallel capabilities of multi-GPU setups. The AdamW optimizer, combined with the cosine annealing strategy, enhances model convergence and mitigates potential overfitting issues during training. To ensure thorough training, an early stopping mechanism is introduced to prevent overfitting on the validation set.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Evaluating indicators</title>
<p>This study employs a set of indicators, such as mAP, Model Size (in MB), Number of Parameters (in MB), and Frames Per Second (FPS). These indicators not only reflect the model&#x2019;s detection accuracy but also provide insights into its computational complexity and real-time performance. The formulas for these indicators are as follows:</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xb7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#xb7;</mml:mo>
<mml:mn>100</mml:mn>
<mml:mo>%</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:msubsup>
<mml:mi>A</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <italic>TP</italic> represents True Positives, <italic>FP</italic> represents False Positives, and <italic>FN</italic> represents False Negatives. Precision measures the proportion of true positive samples among all samples predicted as positive, reflecting the accuracy of the model&#x2019;s positive predictions. Recall, on the other hand, indicates the proportion of true positive samples correctly identified by the model among all actual positive samples, assessing the model&#x2019;s recognition ability. The mean Average Precision (mAP) is calculated by averaging the Average Precision (AP) for each class, with <italic>K</italic> denoting the total number of classes. It serves as a comprehensive performance metric in object detection tasks, demonstrating the model&#x2019;s effectiveness across multiple categories. The F1-score, which is the harmonic mean of Precision and Recall, offers a balanced evaluation of these metrics, particularly useful when there&#x2019;s an imbalance between them.</p>
<p>In addition to these accuracy evaluation metrics, this study incorporates the following indicators to assess model efficiency:</p>
<list list-type="bullet">
<list-item>
<p>
<bold>Model Size and Number of Parameters</bold>: Measured in MB, these metrics reflect the model&#x2019;s complexity and storage requirements. In practical applications, the model&#x2019;s size and parameter count directly influence memory usage and hardware demands, making them crucial for evaluation.</p>
</list-item>
<list-item>
<p>
<bold>Frames Per Second (FPS)</bold>: This metric indicates the number of image frames the model can process per second during operation. A higher FPS signifies greater operational efficiency, which is essential for real-time detection tasks. FPS not only gauges the model&#x2019;s performance on hardware but also its potential for real-time applications in various scenarios.</p>
</list-item>
</list>
<p>Through a comprehensive evaluation of these indicators, this study not only confirms the accuracy and generalization capabilities of the TomatoGuard-YOLO model in tomato disease detection but also examines its computational efficiency and resource consumption, providing robust support for its practical application.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Learning rate selection</title>
<p>The learning rate is a crucial hyperparameter in training deep learning models, significantly influencing convergence speed, final performance, and stability. To identify the optimal learning rate for the TomatoGuard-YOLO model, we designed multiple comparative experiments, testing initial rates of 0.1, 0.05, 0.01, and 0.001 while keeping other hyperparameters constant. <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref> illustrates the trends in the loss function and accuracy for these different learning rates.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Shows Loss and mAP changes of TomatoGuard-YOLO model under different learning rates: <bold>(A)</bold> Loss for Different Learning Rates <bold>(B)</bold> mAP for Different Learning Rates.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g009.tif"/>
</fig>
<p>As illustrated in the loss function curves in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9A</bold>
</xref>, it is clear that with learning rates of 0.1 and 0.05, the loss function does not exhibit substantial reduction, instead displaying a relatively stable or even fluctuating trend. This suggests that the learning rate is excessively high, resulting in overly large gradient update steps that hinder the model&#x2019;s convergence to an optimal solution, thereby affecting training outcomes. Conversely, with learning rates of 0.01 and 0.001, the loss function demonstrates a stable and marked downward trend, indicating that the model gradually approaches the optimal solution. Among these, a learning rate of 0.001 yields the most consistent decrease in the loss function, ultimately reaching its lowest point and demonstrating the best convergence.</p>
<p>
<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9B</bold>
</xref> presents the accuracy changes of the model under different learning rates. It is evident that at learning rates of 0.1 and 0.05, the accuracy curves exhibit significant fluctuations and fail to stabilize at a high level. This aligns with the earlier observation of the loss function&#x2019;s instability, further confirming that a high learning rate induces training volatility, preventing the model from achieving optimal performance. In contrast, at a learning rate of 0.01, the accuracy rises rapidly, but in the mid-to-late training stages, the accuracy curve begins to display slight fluctuations and tendencies towards overfitting. When employing a learning rate of 0.001, accuracy improves steadily, ultimately reaching the highest value without significant overfitting, resulting in a stable training process with excellent convergence.</p>
<p>Considering the convergence behavior of the loss function, the rate of accuracy improvement, and the model&#x2019;s stability during training, we ultimately selected 0.001 as the initial learning rate for the TomatoGuard-YOLO model. This learning rate ensures stable convergence while providing sufficient capacity for the model to fully adapt to the complex patterns and variations in the training data, laying a solid foundation for subsequent optimizations.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Model training</title>
<p>Based on the experimental environment configuration and parameter settings described earlier, we systematically trained the TomatoGuard-YOLO model for a total of 300 epochs. To comprehensively understand the performance during training, we closely monitored the changes in several key performance indicators, such as the loss function, mAP50, and mAP50:95. <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref> shows the trends of these indicators during the training process.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Training process curves.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g010.tif"/>
</fig>
<p>As illustrated in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, both the training loss and validation loss exhibit a consistent and marked decline, with the gap between them gradually narrowing. This suggests that the model is effectively adapting to the training data while simultaneously enhancing its generalization capability. Specifically, the box_loss decreases from an initial value of 0.086 to around 0.030, obj_loss drops from 0.038 to approximately 0.027, and cls_loss significantly reduces from an initial 0.071 to 0.007. These results indicate notable enhancements in the model&#x2019;s localization accuracy (box_loss), detection confidence (obj_loss), and classification performance (cls_loss).</p>
<p>In terms of detection accuracy, the mAP50 and mAP50:95 indicators show a rapid increase in the early training stages, followed by a gradual slowdown and eventual stabilization in the later phases. Ultimately, mAP50 reaches 94.23%, while mAP50:95 stabilizes at 72.52%, indicating that the TomatoGuard-YOLO model possesses excellent object detection capabilities across different IoU thresholds.</p>
<p>Additionally, during the training process, we observed that the model&#x2019;s ability to recognize various types of tomato diseases gradually improved. Notably, it maintained high detection accuracy even for disease types with relatively fewer samples. Through gradual adjustments and optimizations, the TomatoGuard-YOLO model achieved outstanding performance in tomato disease detection.</p>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Ablation study results</title>
<p>To systematically evaluate the effectiveness of our proposed improvements, we conducted comprehensive ablation experiments on the core enhancement modules based on the YOLOv10 model. Through eight carefully designed comparative experiments with different module combinations, we assessed the impact of the MPIRU module, c2f-DFAF module, and Focal-EIoU loss function on tomato disease detection performance across multiple dimensions, including detection accuracy, model size, and computational overhead. <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref> presents detailed key performance indicators under various module combinations.</p>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Ablation experiment results.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">MPIRU</th>
<th valign="middle" align="center">c2f-DFAF</th>
<th valign="middle" align="center">Focal-EIoU</th>
<th valign="middle" align="center">mAP50/%</th>
<th valign="middle" align="center">mAP50:95/%</th>
<th valign="middle" align="center">Model Size/MB</th>
<th valign="middle" align="center">Parameters/MB</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">1</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">84.15</td>
<td valign="middle" align="center">61.22</td>
<td valign="middle" align="center">5.61</td>
<td valign="middle" align="center">2.73</td>
</tr>
<tr>
<td valign="middle" align="center">2</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">91.82</td>
<td valign="middle" align="center">68.71</td>
<td valign="middle" align="center">2.62</td>
<td valign="middle" align="center">0.93</td>
</tr>
<tr>
<td valign="middle" align="center">3</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">88.92</td>
<td valign="middle" align="center">65.81</td>
<td valign="middle" align="center">5.62</td>
<td valign="middle" align="center">2.74</td>
</tr>
<tr>
<td valign="middle" align="center">4</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">88.72</td>
<td valign="middle" align="center">65.61</td>
<td valign="middle" align="center">5.61</td>
<td valign="middle" align="center">2.73</td>
</tr>
<tr>
<td valign="middle" align="center">5</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">91.53</td>
<td valign="middle" align="center">68.42</td>
<td valign="middle" align="center">5.62</td>
<td valign="middle" align="center">2.74</td>
</tr>
<tr>
<td valign="middle" align="center">6</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">93.91</td>
<td valign="middle" align="center">71.23</td>
<td valign="middle" align="center">2.63</td>
<td valign="middle" align="center">0.93</td>
</tr>
<tr>
<td valign="middle" align="center">7</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">93.37</td>
<td valign="middle" align="center">70.69</td>
<td valign="middle" align="center">2.64</td>
<td valign="middle" align="center">0.94</td>
</tr>
<tr>
<td valign="middle" align="center">8</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">&#x221a;</td>
<td valign="middle" align="center">94.23</td>
<td valign="middle" align="center">72.52</td>
<td valign="middle" align="center">2.65</td>
<td valign="middle" align="center">0.94</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The experimental results reveal the significant impact of each enhancement module on model performance. The standalone introduction of the MPIRU module led to substantial improvements in detection accuracy, with mAP50 increasing from 84.15% to 91.82% and mAP50:95 rising from 61.22% to 68.71%. More importantly, this performance enhancement was accompanied by a significant reduction in model complexity, with model size decreasing from 5.61MB to 2.62MB and parameter count reducing from 2.73MB to 0.93MB. These results convincingly demonstrate the MPIRU module&#x2019;s effectiveness in simultaneously improving detection accuracy and achieving model lightweighting.</p>
<p>The integration of the c2f-DFAF module, through optimization of the C2f structure, further enhanced model performance, achieving an mAP50 of 88.92% and mAP50:95 of 65.81%. While the parameter count increased slightly (from 2.73MB to 2.74MB), the significant performance improvements fully justify this optimization. The module demonstrated excellence in enhancing feature hierarchy capture and multi-scale feature fusion, effectively improving the model&#x2019;s feature expression capabilities.</p>
<p>The application of the Focal-EIoU loss function exhibited unique advantages, improving mAP50 to 88.72% and mAP50:95 to 65.61% without increasing model complexity. This enhancement played a crucial role in optimizing bounding box regression precision and addressing sample imbalance issues while improving the model&#x2019;s detection robustness across different target scales.</p>
<p>The synergistic combination of all three enhancement modules (Group 8) achieved optimal performance, with mAP50 reaching 94.23% and mAP50:95 rising to 72.52%, while maintaining a compact model size (2.65MB) and parameter count (0.94MB). Compared to the original YOLOv10 model, detection accuracy improved by 10.07 percentage points, while model size and parameter count reduced by 52.76% and 65.57%, respectively. These results conclusively demonstrate the synergistic effects of the enhancement modules in achieving both high-precision detection and model lightweighting objectives.</p>
<p>The ablation study results deeply reveal the core functions of each module: the MPIRU module serves as the foundation for lightweight design, significantly reducing model complexity; the c2f-DFAF module enhances multi-scale target detection capabilities through feature layer optimization; and the Focal-EIoU loss function optimizes detection accuracy while maintaining low computational overhead. The organic integration of these modules enables TomatoGuard-YOLO to achieve lightweight design while maintaining high performance, providing reliable technical support for resource-constrained practical agricultural application scenarios.</p>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Comparative experiments</title>
<p>In the comparative experiments, we standardized the hyperparameters across all models to ensure a fair comparison. Specifically, all models were trained with the following hyperparameters: a batch size of 64, an initial learning rate of 0.001, the AdamW optimizer, and a cosine annealing learning rate scheduling strategy. Training was conducted for 300 epochs with an early stopping mechanism to prevent overfitting. Additionally, to maintain comparability, all experiments were performed on the same hardware environment (NVIDIA A100 GPU), which maximized the performance of each model while ensuring consistent experimental conditions. To assess the efficacy of the TomatoGuard-YOLO algorithm, we performed comprehensive comparative experiments alongside other state-of-the-art object detection algorithms. The results are summarized in <xref ref-type="table" rid="T9">
<bold>Table&#xa0;9</bold>
</xref>.</p>
<table-wrap id="T9" position="float">
<label>Table&#xa0;9</label>
<caption>
<p>Comparative experiment results.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">mAP50/%</th>
<th valign="middle" align="center">Parameters/MB</th>
<th valign="middle" align="center">Model Size/MB</th>
<th valign="middle" align="center">FPS</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">SSD</td>
<td valign="middle" align="center">63.92</td>
<td valign="middle" align="center">24.16</td>
<td valign="middle" align="center">28.42</td>
<td valign="middle" align="center">89.83</td>
</tr>
<tr>
<td valign="middle" align="center">FasterR-CNN</td>
<td valign="middle" align="center">70.78</td>
<td valign="middle" align="center">136.75</td>
<td valign="middle" align="center">89.64</td>
<td valign="middle" align="center">21.74</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv3</td>
<td valign="middle" align="center">79.23</td>
<td valign="middle" align="center">61.57</td>
<td valign="middle" align="center">123.65</td>
<td valign="middle" align="center">45.26</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv3-tiny</td>
<td valign="middle" align="center">56.52</td>
<td valign="middle" align="center">8.68</td>
<td valign="middle" align="center">17.53</td>
<td valign="middle" align="center">104.15</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv5s</td>
<td valign="middle" align="center">72.73</td>
<td valign="middle" align="center">7.06</td>
<td valign="middle" align="center">14.52</td>
<td valign="middle" align="center">98.67</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOX-s</td>
<td valign="middle" align="center">72.62</td>
<td valign="middle" align="center">9.02</td>
<td valign="middle" align="center">8.95</td>
<td valign="middle" align="center">97.28</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv7-tiny</td>
<td valign="middle" align="center">75.43</td>
<td valign="middle" align="center">6.04</td>
<td valign="middle" align="center">12.32</td>
<td valign="middle" align="center">106.56</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8n</td>
<td valign="middle" align="center">73.42</td>
<td valign="middle" align="center">3.02</td>
<td valign="middle" align="center">6.22</td>
<td valign="middle" align="center">119.45</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv10</td>
<td valign="middle" align="center">84.15</td>
<td valign="middle" align="center">2.73</td>
<td valign="middle" align="center">5.61</td>
<td valign="middle" align="center">125.62</td>
</tr>
<tr>
<td valign="middle" align="center">TomatoGuard-YOLO</td>
<td valign="middle" align="center">94.23</td>
<td valign="middle" align="center">0.94</td>
<td valign="middle" align="center">2.65</td>
<td valign="middle" align="center">129.64</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The comparative results presented in <xref ref-type="table" rid="T9">
<bold>Table&#xa0;9</bold>
</xref> clearly demonstrate that the TomatoGuard-YOLO model achieves a significant advantage in the mAP50 metric, reaching 94.23%, which is far superior to other compared algorithms. Especially compared to the original YOLOv10, TomatoGuard-YOLO improves by 10.07 percentage points, fully demonstrating the efficacy of our improvement strategies.</p>
<p>As for model complexity, TomatoGuard-YOLO&#x2019;s parameter count is only 0.94M, a reduction of 65.57% compared to YOLOv10&#x2019;s 2.73M. The model size is 2.65M, a reduction of 52.76% compared to YOLOv10&#x2019;s 5.61M. This indicates that while improving detection accuracy, TomatoGuard-YOLO significantly reduces model complexity, making it a better choice for deployment in resource-limited environments.</p>
<p>Regarding inference speed, TomatoGuard-YOLO achieves an average frame rate of 129.64 FPS, an improvement of 3.20% compared to YOLOv10&#x2019;s 125.62 FPS. This further demonstrates that despite the significant performance improvements, the model does not sacrifice inference speed; instead, it enhances inference efficiency. This high-efficiency and lightweight characteristic makes TomatoGuard-YOLO perform exceptionally well.</p>
<p>To further assess the benefits of TomatoGuard-YOLO, we performed a comparison of mAP and Loss curves with other models, as shown in <xref ref-type="fig" rid="f11">
<bold>Figures&#xa0;11A, B</bold>
</xref>.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Shows Comparison of mAP and loss curves for different models: <bold>(A)</bold> mAP0.5 Comparison of 10 Models <bold>(B)</bold> Loss Comparison of 10 Models.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g011.tif"/>
</fig>
<p>From <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11A</bold>
</xref>, the mAP curve for TomatoGuard-YOLO is notably superior to those of the other models under comparison, ultimately stabilizing at approximately 94.23%, which reinforces the model&#x2019;s strong performance in tomato disease detection. <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11B</bold>
</xref> illustrates that the initial loss for TomatoGuard-YOLO is 0.195359, significantly lower than that of the other models, eventually stabilizing at around 0.0626034. This suggests that the model achieves a quicker convergence and a lower final loss, further confirming its optimization efficacy. Overall, the outstanding performance of TomatoGuard-YOLO regarding accuracy, speed, and compact design highlights its significant potential for application in tomato disease detection within complex environments.</p>
</sec>
<sec id="s4_7">
<label>4.7</label>
<title>Detection results for different types of diseases</title>
<p>To thoroughly assess the TomatoGuard-YOLO model&#x2019;s effectiveness in identifying different tomato diseases, we evaluated it using a custom imbalanced dataset. The detection results for various disease types, as achieved by the TomatoGuard-YOLO model, are summarized in <xref ref-type="table" rid="T10">
<bold>Table&#xa0;10</bold>
</xref>.</p>
<table-wrap id="T10" position="float">
<label>Table&#xa0;10</label>
<caption>
<p>Detection performance of the TomatoGuard-YOLO model across various disease types.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="bottom" align="center">Category</th>
<th valign="bottom" align="center">Precision (%)</th>
<th valign="bottom" align="center">Recall (%)</th>
<th valign="bottom" align="center">F1-score (%)</th>
<th valign="bottom" align="center">AP50 (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">Healthy</td>
<td valign="bottom" align="center">96.85</td>
<td valign="bottom" align="center">96.53</td>
<td valign="bottom" align="center">96.69</td>
<td valign="bottom" align="center">97.59</td>
</tr>
<tr>
<td valign="bottom" align="center">Late Blight</td>
<td valign="bottom" align="center">95.93</td>
<td valign="bottom" align="center">95.41</td>
<td valign="bottom" align="center">95.67</td>
<td valign="bottom" align="center">95.99</td>
</tr>
<tr>
<td valign="bottom" align="center">Early Blight</td>
<td valign="bottom" align="center">94.72</td>
<td valign="bottom" align="center">94.28</td>
<td valign="bottom" align="center">94.51</td>
<td valign="bottom" align="center">94.11</td>
</tr>
<tr>
<td valign="bottom" align="center">Gray Mold</td>
<td valign="bottom" align="center">93.16</td>
<td valign="bottom" align="center">92.58</td>
<td valign="bottom" align="center">92.87</td>
<td valign="bottom" align="center">93.07</td>
</tr>
<tr>
<td valign="bottom" align="center">Leaf Mold</td>
<td valign="bottom" align="center">91.74</td>
<td valign="bottom" align="center">90.86</td>
<td valign="bottom" align="center">91.32</td>
<td valign="bottom" align="center">90.36</td>
</tr>
<tr>
<td valign="bottom" align="center">Average</td>
<td valign="bottom" align="center">94.48</td>
<td valign="bottom" align="center">93.93</td>
<td valign="bottom" align="center">94.21</td>
<td valign="bottom" align="center">94.23</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="table" rid="T10">
<bold>Table&#xa0;10</bold>
</xref> demonstrates that the TomatoGuard-YOLO model attains notable Precision, Recall, and AP metrics above 90% for all four disease types and healthy samples, demonstrating high accuracy and recall rates. The model&#x2019;s mAP reaches 94.23%, fully proving its excellent performance in handling different types of tomato diseases. Notably, the model excels not only in detecting common diseases such as late blight and early blight but also maintains high detection sensitivity for relatively rare diseases like gray mold and leaf mold. Additionally, the model&#x2019;s outstanding performance in recognizing healthy samples helps reduce misdiagnosis and unnecessary treatments, which is crucial for practical agricultural production.</p>
<p>The AP50 values for detecting five types of tomato diseases and healthy samples using the proposed TomatoGuard-YOLO algorithm and other models are displayed in <xref ref-type="table" rid="T11">
<bold>Table&#xa0;11</bold>
</xref>. The findings clearly demonstrate that the TomatoGuard-YOLO algorithm shows enhanced adaptability in handling targets with pronounced sample imbalance and size variations.</p>
<table-wrap id="T11" position="float">
<label>Table&#xa0;11</label>
<caption>
<p>Comparison of AP50 results for different models in tomato disease detection tasks.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Method</th>
<th valign="middle" rowspan="2" align="center">mAP50/%</th>
<th valign="middle" colspan="5" align="center">AP50/%</th>
</tr>
<tr>
<th valign="middle" align="center">Healthy</th>
<th valign="middle" align="center">Late Blight</th>
<th valign="middle" align="center">Early Blight</th>
<th valign="middle" align="center">Gray Mold</th>
<th valign="middle" align="center">Leaf Mold</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">SSD</td>
<td valign="middle" align="center">66.56</td>
<td valign="middle" align="center">71.52</td>
<td valign="middle" align="center">69.73</td>
<td valign="middle" align="center">65.84</td>
<td valign="middle" align="center">62.82</td>
<td valign="middle" align="center">62.89</td>
</tr>
<tr>
<td valign="middle" align="center">FasterR-CNN</td>
<td valign="middle" align="center">73.56</td>
<td valign="middle" align="center">79.04</td>
<td valign="middle" align="center">77.07</td>
<td valign="middle" align="center">72.77</td>
<td valign="middle" align="center">69.44</td>
<td valign="middle" align="center">69.51</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv3</td>
<td valign="middle" align="center">82.32</td>
<td valign="middle" align="center">88.45</td>
<td valign="middle" align="center">86.24</td>
<td valign="middle" align="center">81.43</td>
<td valign="middle" align="center">77.70</td>
<td valign="middle" align="center">77.78</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv3-tiny</td>
<td valign="middle" align="center">58.68</td>
<td valign="middle" align="center">63.05</td>
<td valign="middle" align="center">61.47</td>
<td valign="middle" align="center">58.04</td>
<td valign="middle" align="center">55.38</td>
<td valign="middle" align="center">55.44</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv5s</td>
<td valign="middle" align="center">76.19</td>
<td valign="middle" align="center">81.87</td>
<td valign="middle" align="center">79.82</td>
<td valign="middle" align="center">75.37</td>
<td valign="middle" align="center">71.92</td>
<td valign="middle" align="center">71.99</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOX-s</td>
<td valign="middle" align="center">75.32</td>
<td valign="middle" align="center">80.93</td>
<td valign="middle" align="center">78.90</td>
<td valign="middle" align="center">74.50</td>
<td valign="middle" align="center">71.09</td>
<td valign="middle" align="center">71.16</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv7-tiny</td>
<td valign="middle" align="center">78.82</td>
<td valign="middle" align="center">84.69</td>
<td valign="middle" align="center">82.57</td>
<td valign="middle" align="center">77.97</td>
<td valign="middle" align="center">74.40</td>
<td valign="middle" align="center">74.47</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv8n</td>
<td valign="middle" align="center">83.20</td>
<td valign="middle" align="center">89.39</td>
<td valign="middle" align="center">87.16</td>
<td valign="middle" align="center">82.30</td>
<td valign="middle" align="center">78.53</td>
<td valign="middle" align="center">78.61</td>
</tr>
<tr>
<td valign="middle" align="center">YOLOv10</td>
<td valign="middle" align="center">84.15</td>
<td valign="middle" align="center">90.71</td>
<td valign="middle" align="center">87.79</td>
<td valign="middle" align="center">83.72</td>
<td valign="middle" align="center">81.37</td>
<td valign="middle" align="center">77.18</td>
</tr>
<tr>
<td valign="middle" align="center">TomatoGuard-YOLO</td>
<td valign="middle" align="center">94.23</td>
<td valign="middle" align="center">97.59</td>
<td valign="middle" align="center">95.99</td>
<td valign="middle" align="center">94.11</td>
<td valign="middle" align="center">93.07</td>
<td valign="middle" align="center">90.36</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T11">
<bold>Table&#xa0;11</bold>
</xref>, the YOLOv10 algorithm achieves an AP50 of 90.71% for detecting healthy samples but only 77.18% for identifying leaf mold. Similarly, the YOLOv8n algorithm performs relatively well for late blight detection, with an AP50 of 87.16%, but struggles with gray mold, achieving only 78.53%. In contrast, the proposed TomatoGuard-YOLO algorithm demonstrates exceptional performance across all categories, significantly surpassing other models in the overall detection of five types of tomato diseases as well as healthy samples. Notably, for the more challenging cases of gray mold and leaf mold, TomatoGuard-YOLO achieves AP50 values of 93.07% and 90.36%, respectively, outperforming competing algorithms by a substantial margin.</p>
<p>Accurate identification and localization of tomato diseases are critical metrics for evaluating detection performance. While healthy samples and late blight are relatively easier to detect due to their distinct features, diseases such as early blight, gray mold, and leaf mold pose greater challenges. These diseases often exhibit subtle symptoms, with blurred boundaries between lesions and healthy tissue, making detection more difficult. Gray mold and leaf mold, in particular, are characterized by irregular lesion distribution, significant variations in lesion size, and high visual similarity to the background, further complicating accurate detection.</p>
<p>Compared to the next-best performer, YOLOv10, TomatoGuard-YOLO achieves a remarkable improvement of 10.08 percentage points in mAP50, with category-specific gains ranging from 6.88 to 13.18 percentage points. This comprehensive performance enhancement highlights the superiority of TomatoGuard-YOLO in tackling the complexities of tomato disease detection. Furthermore, it provides reliable technical support for precise diagnosis and timely intervention, offering significant practical value in real-world agricultural applications.</p>
</sec>
<sec id="s4_8">
<label>4.8</label>
<title>Visualization of disease detection results</title>
<p>To visually compare the detection capabilities of the TomatoGuard-YOLO model, <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref> highlights the performance differences between YOLOv10 and TomatoGuard-YOLO in representative tomato disease scenarios.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Detection results comparison in typical Tomato disease scenarios <bold>(A)</bold> Ground Truth, <bold>(B)</bold> Baseline, and <bold>(C)</bold> TomatoGuard-YOLO.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1499278-g012.tif"/>
</fig>
<p>The visual results clearly demonstrate that TomatoGuard-YOLO outperforms YOLOv10, particularly in handling complex backgrounds and multi-scale targets. These advantages underscore the robustness and precision of TomatoGuard-YOLO in accurately identifying and localizing diseased regions under challenging conditions, further validating its superiority in practical agricultural applications. In detecting small targets, TomatoGuard-YOLO demonstrates extremely high precision, especially in identifying early disease spots, accurately locating lesions. This is crucial for timely detection and control measures. In complex situations such as overlapping and occluded leaves, TomatoGuard-YOLO can still effectively distinguish disease areas, significantly reducing false detection rates, showing high robustness in complex scenarios. Additionally, in uneven lighting or complex shadow backgrounds, TomatoGuard-YOLO exhibits excellent environmental adaptability, accurately identifying disease areas, ensuring detection stability under different lighting conditions. Compared to YOLOv10, TomatoGuard-YOLO generates more precise bounding boxes, aiding in accurately assessing the severity and spread of diseases, providing more reliable support for actual disease assessment. Therefore, by introducing innovations such as MPIRU, DFAF, and Focal-EIoU loss functions, TomatoGuard-YOLO significantly improves detection accuracy and robustness while maintaining model lightweight, excelling in multi-scale and complex scenarios of tomato disease detection. Moreover, the model shows outstanding detection performance in quantitative evaluation metrics, providing strong technical support for early warning and precise control of tomato diseases in practical applications.</p>
</sec>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions and future directions</title>
<sec id="s5_1" sec-type="conclusions">
<label>5.1</label>
<title>Conclusion</title>
<p>This research introduces a highly efficient and lightweight object detection approach built on an enhanced version of YOLOv10 for the challenging task of detecting tomato diseases&#x2014;TomatoGuard-YOLO. By implementing an innovative model architecture and conducting thorough experimental evaluations, notable outcomes have been achieved. The key conclusions are as follows:</p>
<list list-type="order">
<list-item>
<p>Incorporating the Multi-Path Inverted Residual Unit (MPIRU) greatly strengthens the model&#x2019;s capacity for multi-scale feature extraction and integration. Experimental findings demonstrate that MPIRU not only boosts detection accuracy but also lowers model complexity, ensuring effective lightweight detection.</p>
</list-item>
<list-item>
<p>The Dynamic Focusing Attention Framework (DFAF) improves the model&#x2019;s precision in identifying critical disease regions. With the C2f module optimized, C2f-DFAF efficiently captures disease features in complex environments, significantly enhancing detection performance with negligible added computational cost.</p>
</list-item>
<list-item>
<p>The Focal-EIoU loss function effectively tackles challenges related to sample imbalance and bounding box regression accuracy, leading to significant improvements in detecting small objects and boundary diseases. In addition to optimizing detection precision, the overall model performance is further improved.</p>
</list-item>
</list>
<p>Experimental results indicate that TomatoGuard-YOLO achieves better performance than current methods on the tomato disease dataset. The model achieves an mAP50 of 94.23%, an improvement of 10.07 percentage points over the original YOLOv10, with a model size reduction of 52.76%, parameter count reduction of 65.31%, and an average inference speed increase to 129.64 FPS. These data fully demonstrate the method&#x2019;s outstanding advantages in model accuracy, efficiency, and lightweight design.</p>
<p>Comparative experiments and visualization results further validate TomatoGuard-YOLO&#x2019;s excellent performance in complex scenarios. Whether in small target detection, differentiation in complex backgrounds, multi-category disease recognition, or bounding box precision control, the proposed model shows significant improvement. These improvements not only provide reliable technical support for disease detection but also lay a solid foundation for applications in practical agricultural scenarios.</p>
<p>In summary, the TomatoGuard-YOLO model demonstrates excellent performance and broad adaptability in tomato disease detection tasks. Its high accuracy and reliability in disease detection provide powerful technical tools for disease prevention and intelligent management in tomato cultivation, offering significant application value in improving tomato cultivation efficiency and reducing disease losses.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Research limitations and future directions</title>
<p>Despite the significant achievements of TomatoGuard-YOLO in tomato disease detection, the current study lacks large-scale field deployment validation, which is crucial for understanding real-world performance. Future research can further explore the following aspects:</p>
<list list-type="order">
<list-item>
<p>Further optimize the design of MPIRU and C2f-DFAF modules, deeply study more efficient feature extraction and attention mechanisms.</p>
</list-item>
<list-item>
<p>Another important research direction is to expand the application scope of the model, exploring its generalization ability in detecting diseases of different crops, and testing its detection effectiveness in more complex field scenarios to evaluate its robustness and adaptability.</p>
</list-item>
<list-item>
<p>Combining edge computing technology, research on deployment optimization strategies for the model on resource-constrained devices is also key to future development. By optimizing deployment on low-power devices, computational resource consumption can be effectively reduced, further enhancing the model&#x2019;s practical application value.</p>
</list-item>
<list-item>
<p>TomatoGuard-YOLO can work with other agricultural intelligent systems (drone, IoT devices, etc.), fully leveraging the complementary advantages of multi-dimensional data to achieve more comprehensive crop health management in actual agricultural environments. This integration can effectively improve early warning capabilities for diseases, promoting the development of intelligent and precise agriculture.</p>
</list-item>
</list>
</sec>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>JL: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. XW: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This research was funded by the Shandong Province Natural Science Foundation (Grant Nos. ZR2023MF048, ZR2023QC116 &amp; ZR2021QC173), the Key Research and Development Program of Shandong Province (Grant No. 2024RZB0206), the Disciplinary Construction Funds of Weifang University of Science and Technology, the Supporting Construction Funds for Shandong Province Data Open Innovation Application Laboratory, the School-level Talent Project (Grant No. 2018RC002), the Weifang Soft Science Project (Grant No. 2023RKX184), and the Weifang City Science and Technology Development Plan Project (Grant No. 2023GX051).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alif</surname> <given-names>M. A. R.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>YOLOv1 to YOLOv10: A comprehensive review of YOLO variants and their application in the agricultural domain</article-title>. <source>arXiv. preprint. arXiv:2406.10139</source>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barbedo</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Factors influencing the use of deep learning for plant disease recognition</article-title>. <source>Biosyst. Eng.</source> <volume>172</volume>, <fpage>84</fpage>&#x2013;<lpage>91</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2018.05.013</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhattacharya</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Somayaji</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Gadekallu</surname> <given-names>T. R.</given-names>
</name>
<name>
<surname>Alazab</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Maddikunta</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A review on deep learning for future smart cities</article-title>. <source>Internet Technol. Lett.</source> <volume>1)</volume>, <fpage>5</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/itl2.v5.1</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C. Y.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H. Y. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>YOLOv4: Optimal speed and accuracy of object detection</article-title>. <source>arXiv. preprint. arXiv:2004.10934</source>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bora</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Parasar</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Charhate</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A detection of tomato plant diseases using deep learning MNDLNN classifier</article-title>. <source>Signal. Image. Video. Process.</source> <volume>17</volume>, <fpage>3255</fpage>&#x2013;<lpage>3263</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11760-023-02498-y</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buja</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Sabella</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Monteduro</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Chiriac&#xf2;</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>Maruccio</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Advances in plant disease detection and monitoring: From traditional assays to in-field diagnostics</article-title>. <source>Sensors</source> <volume>21</volume>, <fpage>2129</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s21062129</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Camargo</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>J. S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>An image-processing based algorithm to automatically identify plant disease visual symptoms</article-title>. <source>Biosyst. Eng</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2008.09.030</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Hza</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
<collab>colleagues</collab>
</person-group> (<year>2021</year>). <article-title>EFDet: An efficient detection method for cucumber disease under natural complex environments</article-title>. <source>Comput. Electron. Agric.</source> <volume>189</volume>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dananjayan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhuang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hou</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Assessment of state-of-the-art deep learning based citrus disease detection techniques using annotated optical leaf images</article-title>. <source>Comput. Electron. Agric.</source> <volume>193</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2021.106658</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Du</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>BLP-YOLOv10: efficient safety helmet detection for low-light mining</article-title>. <source>J. Real-Time. Image. Process.</source> <volume>22</volume>, <elocation-id>10</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11554-024-01587-6</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fuentes</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Yoon</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>S. C.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>D. S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A robust deep-learning-based detector for real-time tomato plant diseases and pests recognition</article-title>. <source>Sensors</source> <volume>17</volume>, <fpage>2022</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s17092022</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fuentes</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Yoon</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>D. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Open set self and across domain adaptation for tomato disease recognition with deep learning techniques</article-title>. <source>Front. Plant Sci</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2021.758027</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ge</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>YOLOX: exceeding YOLO series in 2021</article-title>. <source>arXiv. preprint. arXiv:2107.08430</source>.</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Fast R-CNN</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE International Conference on Computer Vision</conf-name>. <fpage>1440</fpage>&#x2013;<lpage>1448</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Squeeze-and-excitation networks</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>7132</fpage>&#x2013;<lpage>7141</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jocher</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Chaurasia</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Stoken</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Borovec</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Kwon</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Michael</surname> <given-names>K.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <source>ultralytics/yolov5: v6. 2-yolov5 classification models, apple m1, reproducibility, clearml and deci.ai integrations</source> (<publisher-name>Zenodo</publisher-name>).</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kadry</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Early detection and classification of tomato leaf disease using high-performance deep neural network</article-title>. <source>Sensors</source> <volume>21</volume>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamilaris</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Prenafeta-Bold&#xfa;</surname> <given-names>F. X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep learning in agriculture: A survey</article-title>. <source>Comput. Electron. Agric.</source> <volume>147</volume>, <fpage>70</fpage>&#x2013;<lpage>90</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2018.02.016</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Toward real scenery: A lightweight tomato growth inspection algorithm for leaf disease detection and fruit counting</article-title>. <source>Plant Phenomics.</source> <volume>6</volume>, <fpage>0174</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/plantphenomics.0174</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaniyassery</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Goyal</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Thorat</surname> <given-names>S. A.</given-names>
</name>
<name>
<surname>Rao</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Chandrashekar</surname> <given-names>H. K.</given-names>
</name>
<name>
<surname>Murali</surname> <given-names>T. S.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Association of meteorological variables with leaf spot and fruit rot disease incidence in eggplant and YOLOv8-based disease classification</article-title>. <source>Ecol. Inf.</source>, <fpage>102809</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2024.102809</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname> <given-names>Y. S.</given-names>
</name>
<name>
<surname>Patil</surname> <given-names>M. P.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>J. G.</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>S. S.</given-names>
</name>
<name>
<surname>Seo</surname> <given-names>Y. B.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>G. D.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Improved tomato leaf disease recognition based on the YOLOv5m with various soft attention module combinations</article-title>. <source>Agriculture</source> <volume>14</volume>, <fpage>1472</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture14091472</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Weng</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Geng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>YOLOv6: A single-stage object detection framework for industrial applications</article-title>. <source>arXiv. preprint. arXiv:2209.02976</source>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Quan</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>High-throughput plant phenotyping platform (HT3P) as a novel tool for estimating agronomic traits from&#xa0;the lab to the field</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>8</volume>, <elocation-id>623705</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fbioe.2020.623705</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Paul</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Tis</surname> <given-names>T. B.</given-names>
</name>
<collab>colleagues</collab>
</person-group> (<year>2019</year>). <article-title>Non-invasive plant disease diagnostics enabled by smartphone-based fingerprinting of leaf volatiles</article-title>. <source>Nat. Plants</source> <volume>5</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41477-019-0476-y</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>T. Y.</given-names>
</name>
<name>
<surname>Goyal</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Doll&#xe1;r</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Focal loss for dense object detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE International Conference on Computer Vision</conf-name>. <fpage>2980</fpage>&#x2013;<lpage>2988</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>He</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Identification of apple leaf diseases based on deep convolutional neural networks</article-title>. <source>Symmetry</source> <volume>10</volume>, <fpage>11</fpage>.</citation>
</ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Anguelov</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Erhan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Szegedy</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Reed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>C. Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). &#x201c;<article-title>SSD: Single shot multibox detector</article-title>,&#x201d; in <conf-name>Computer Vision&#x2013;ECCV 2016: 14th European Conference</conf-name>. <fpage>21</fpage>&#x2013;<lpage>37</lpage> (<publisher-name>Springer</publisher-name>).</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Shu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Hancke</surname> <given-names>G. P.</given-names>
</name>
<name>
<surname>Abu-Mahfouz</surname> <given-names>A. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>From industry 4.0 to agriculture 4.0: Current status, enabling technologies, and research challenges</article-title>. <source>IEEE Trans. Ind. Inf.</source> <volume>pp</volume>, <fpage>1</fpage>&#x2013;<lpage>1</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lowe</surname> <given-names>D. G.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Distinctive image features from scale-invariant keypoints</article-title>. <source>Int. J. Comput. Vision</source> <volume>60</volume>, <fpage>91</fpage>&#x2013;<lpage>110</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1023/B:VISI.0000029664.99615.94</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ma</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>H. T.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>ShuffleNet V2: Practical guidelines for efficient CNN architecture design</article-title>,&#x201d; in <conf-name>Proceedings of the European Conference on Computer Vision (ECCV)</conf-name>. <fpage>116</fpage>&#x2013;<lpage>131</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mei</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>BGF-YOLOv10: small object detection algorithm from unmanned aerial vehicle perspective based on improved YOLOv10</article-title>. <source>Sensors</source> <volume>24</volume>, <elocation-id>6911</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s24216911</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qi</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>K.</given-names>
</name>
<collab>colleagues</collab>
</person-group> (<year>2022</year>). <article-title>An improved YOLOv5s model based on visual attention mechanism: Application to recognition of tomato virus disease</article-title>. <source>Comput. Electron. Agric.</source> <volume>194</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.106780</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Divvala</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>You only look once: Unified, real-time object detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>779</fpage>&#x2013;<lpage>788</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saleem</surname> <given-names>M. H.</given-names>
</name>
<name>
<surname>Potgieter</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Arif</surname> <given-names>K. M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Automation in agriculture by machine and deep learning techniques: A review of recent developments</article-title>. <source>Precis. Agric.</source> <volume>6)</volume>, <fpage>22</fpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sandler</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Howard</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zhmoginov</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>L. C.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>MobileNetV2: Inverted residuals and linear bottlenecks</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>4510</fpage>&#x2013;<lpage>4520</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sangaiah</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>F. N.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Y. B.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>W. C.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <source>UAV T-YOLO-Rice: An enhanced tiny YOLO networks for rice leaves diseases detection in paddy agronomy</source> (<publisher-name>IEEE Transactions on Network Science and Engineering</publisher-name>).</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shekhawat</surname> <given-names>R. S.</given-names>
</name>
<name>
<surname>Sinha</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Review of image processing approaches for detecting plant diseases</source> (<publisher-name>IET Image Processing</publisher-name>).</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ganapathysubramanian</surname> <given-names>B.</given-names>
</name>
<collab>colleagues</collab>
</person-group> (<year>2020</year>). <article-title>Challenges and opportunities in machine-augmented plant stress phenotyping</article-title>. <source>Trends Plant Sci</source>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>B.</given-names>
</name>
<collab>colleagues</collab>
</person-group> (<year>2021</year>). <article-title>MEAN-SSD: A novel real-time detector for apple leaf diseases using improved light-weight convolutional neural networks</article-title>. <source>Comput. Electron. Agric.</source> <volume>189</volume>, <fpage>106379</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2021.106379</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sunil</surname> <given-names>C. K.</given-names>
</name>
<name>
<surname>Jaidhar</surname> <given-names>C. D.</given-names>
</name>
<name>
<surname>Patil</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Tomato plant disease classification using multilevel feature fusion with adaptive channel spatial and pixel attention mechanism</article-title>. <source>Expert Syst. Appl.</source> <volume>228</volume>, <fpage>120381</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2023.120381</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>YOLOv10: Real-time end-to-end object detection</article-title>. <source>arXiv. preprint. arXiv:2405.14458</source>.</citation>
</ref>
<ref id="B42">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C. Y.</given-names>
</name>
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H. Y. M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>. <fpage>7464</fpage>&#x2013;<lpage>7475</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Automatic image-based plant disease severity estimation using deep learning</article-title>. <source>Comput. Intell. Neurosci.</source> <volume>2017</volume>, <fpage>2917536</fpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wiesner-Hanks</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Stewart</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Dechant</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Nelson</surname> <given-names>R. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Millimeter-level plant disease detection from aerial photographs via deep learning and crowdsourced data</article-title>. <source>Front. Plant Sci.</source> <volume>10</volume>, <elocation-id>1550</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2019.01550</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Woo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J. Y.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>CBAM: Convolutional block attention module</article-title>,&#x201d; in <conf-name>Proceedings of the European Conference on Computer Vision (ECCV)</conf-name>. <fpage>3</fpage>&#x2013;<lpage>19</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Qiu</surname> <given-names>F.</given-names>
</name>
<collab>colleagues</collab>
</person-group> (<year>2021</year>). <article-title>Detecting soybean leaf disease from synthetic image using multi-feature fusion faster R-CNN</article-title>. <source>Comput. Electron. Agric.</source> <volume>183</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2021.106064</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Can deep learning identify tomato leaf disease</article-title>? <source>Adv. Multimedia.</source> <volume>2018</volume>, <fpage>6710865</fpage>.</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Identification of tomato leaf diseases based on multi-channel automatic orientation recurrent attention network</article-title>. <source>Comput. Electron. Agric.</source> <volume>205</volume>, <fpage>107605</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2022.107605</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>EIoU: A new loss function for bounding box regression</article-title>,&#x201d; in <conf-name>Proceedings of the AAAI Conference on Artificial Intelligence</conf-name>, Vol. <volume>35</volume>. <fpage>3188</fpage>&#x2013;<lpage>3195</lpage>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Tomato leaf disease identification based on convolutional neural network</article-title>. <source>Comput. Electron. Agric.</source> <volume>167</volume>, <fpage>105075</fpage>.</citation>
</ref>
</ref-list>
</back>
</article>