<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1611865</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Intelligent deep learning architecture for precision vegetable disease detection advancing agricultural new quality productive forces</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Jun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/694621/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Xuewei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3063003/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chen</surname>
<given-names>Qian</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yan</surname>
<given-names>Peng</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Guo</surname>
<given-names>Dugang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Shandong Provincial University Laboratory for Protected Horticulture, Weifang University of Science and Technology</institution>, <addr-line>Weifang</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Computer, Sichuan Technology and Business University</institution>, <addr-line>Chengdu, Sichuan</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>The Industry-Education Integration Office, Sichuan Technology and Business University</institution>, <addr-line>Chengdu, Sichuan</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Anirban Roy, Indian Council of Agricultural Research (ICAR), India</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Giao Nguyen, Department of Primary Industries and Regional Development of Western Australia (DPIRD), Australia</p>
<p>Meena Pandey, University of California, Davis, United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Qian Chen, <email xlink:href="mailto:chenqianwork2019@163.com">chenqianwork2019@163.com</email>; Peng Yan, <email xlink:href="mailto:nic@stbu.edu.cn">nic@stbu.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1611865</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Liu, Wang, Chen, Yan and Guo.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Liu, Wang, Chen, Yan and Guo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>In the context of advancing agricultural new quality productive forces, addressing the challenges of uneven illumination, target occlusion, and mixed infections in greenhouse vegetable disease detection becomes crucial for modern precision agriculture. To tackle these challenges, this study proposes YOLO-vegetable, a high-precision detection algorithm based on improved You Only Look Once version 10 (YOLOv10). The framework incorporates three innovative modules. The Adaptive Detail Enhancement Convolution (ADEConv) module employs dynamic parameter adjustment to preserve fine-grained features while maintaining computational efficiency. The Multi-granularity Feature Fusion Detection Layer (MFLayer) improves small target localization accuracy through cross-level feature interaction mechanisms. The Inter-layer Dynamic Fusion Pyramid Network (IDFNet) combines with Attention-guided Adaptive Feature Selection (AAFS) mechanism to enhance key information extraction capability. Experimental validation on our self-built Vegetable Disease Dataset (VDD, 15,000 images) demonstrates that YOLO-vegetable achieves 95.6% mean Average Precision at IoU threshold 0.5, representing a 6.4 percentage point improvement over the baseline model. The method maintains efficiency with 3.8M parameters and 18.6ms inference time per frame, providing a practical solution for intelligent disease detection in facility agriculture and contributing to the development of agricultural new quality productive forces.</p>
</abstract>
<kwd-group>
<kwd>agricultural new quality productive forces</kwd>
<kwd>deep learning</kwd>
<kwd>vegetable disease detection</kwd>
<kwd>YOLO</kwd>
<kwd>precision agriculture</kwd>
<kwd>greenhouse cultivation</kwd>
<kwd>attention mechanism</kwd>
</kwd-group>
<counts>
<fig-count count="15"/>
<table-count count="6"/>
<equation-count count="7"/>
<ref-count count="45"/>
<page-count count="20"/>
<word-count count="9309"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>With the intensification of global population growth and climate change challenges, developing new quality productive forces in agriculture has become a strategic choice for ensuring food security and promoting sustainable agricultural development. New quality productive forces in agriculture emphasize the construction of efficient, green, and sustainable modern agricultural production systems through technological innovation, digital transformation, and intelligent upgrading. Against this backdrop, intelligent agricultural disease detection and recognition technology, as a core component of digital agriculture, is becoming a key technological support for driving agricultural productivity transformation.</p>
<p>Intelligent detection and recognition of agricultural diseases is a key technology for ensuring agricultural production and food security. With the rapid development of facility agriculture, greenhouse cultivation has become an important mode of modern agricultural production, representing a typical application of new quality productive forces in facility agriculture. Although greenhouse environments provide better disease control conditions compared to open fields, the enclosed conditions and high plant density can still facilitate rapid disease transmission when outbreaks occur, making early and accurate detection crucial for preventing significant yield losses. Statistics show that greenhouse vegetable diseases alone cause 20-30% global yield losses annually (<xref ref-type="bibr" rid="B37">W&#xf3;jcik Gront et&#xa0;al., 2024</xref>). Traditional manual inspection methods are inefficient and susceptible to subjective factors in complex greenhouse environments, making it difficult to meet the monitoring needs of large-scale facility agriculture, urgently requiring revolutionary changes in detection methods through artificial intelligence technology.</p>
<p>Deep learning, particularly Convolutional Neural Networks (CNNs), has revolutionized computer vision with excellent feature extraction capabilities (<xref ref-type="bibr" rid="B11">Chowdhury et&#xa0;al., 2020</xref>). Classic architectures like VGG and ResNet show strong performance in disease recognition, while recent object detection advances provide new pathways for intelligent disease detection (<xref ref-type="bibr" rid="B6">Bonora et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B4">Bao et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B24">Mathieu et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B16">Jian et&#xa0;al., 2025</xref>). Recent advancements in Vision Transformers (ViTs), such as CrossViT (<xref ref-type="bibr" rid="B10">Chen et&#xa0;al., 2021</xref>) and DaViT (<xref ref-type="bibr" rid="B12">Ding et&#xa0;al., 2022</xref>), have demonstrated strong performance in image classification tasks. However, in greenhouse vegetable disease detection, these transformer-based architectures face significant limitations. Their computational complexity scales quadratically with input resolution, making them resource-intensive for real-time applications. While transformers excel at capturing global context, they often struggle with the fine-grained features essential for identifying small disease lesions under variable greenhouse lighting and occlusion conditions. Our proposed YOLO-vegetable model addresses these limitations through adaptive convolutional mechanisms specifically optimized for greenhouse environments.</p>
<p>Among various deep learning architectures, YOLO (You Only Look Once) series networks have become important models for disease detection in greenhouse environments due to their excellent real-time performance and detection accuracy. The YOLO series object detection algorithms have continuously evolved since their introduction in 2016, experiencing multiple significant upgrades from YOLOv1 to YOLOv10, achieving remarkable progress in detection accuracy, real-time performance, and resource consumption (<xref ref-type="bibr" rid="B3">Alif and Hussain, 2024</xref>). The recently proposed YOLOv10 further improves model detection performance in complex scenarios through optimized backbone network architecture and feature extraction strategies (<xref ref-type="bibr" rid="B35">Wang et&#xa0;al., 2024</xref>). However, existing YOLO variants face fundamental limitations in greenhouse applications due to three critical gaps: feature preservation challenges during downsampling operations, inadequate multi-scale adaptation for disease manifestations ranging from macro-level patterns to micro-level changes, and lack of dynamic feature fusion mechanisms for varying greenhouse environmental conditions.</p>
<p>YOLOv10 was selected as our baseline architecture for several key reasons: (1) It represents the latest advancement in the YOLO series with optimized dual-head design eliminating non-maximum suppression during inference, reducing computational overhead; (2) YOLOv10n provides the optimal balance between parameter efficiency (2.2M parameters) and detection capability, making it suitable for resource-constrained agricultural deployment scenarios; (3) Its backbone architecture incorporates modern design principles including attention mechanisms and efficient feature extraction, providing a solid foundation for our agricultural-specific modifications; (4) Extensive benchmarking shows YOLOv10 outperforms YOLOv8 and earlier versions in both accuracy and inference speed, establishing it as the current state-of-the-art for real-time object detection applications.</p>
<p>Vegetable disease detection in greenhouse environments faces several unique challenges. Although greenhouse environments provide more stable and controllable conditions compared to open fields, computer vision systems must still handle varying lighting conditions due to natural light changes throughout the day, reflections and scattering caused by glass or film covering materials, and shadows created by structural elements, all of which can affect image quality and detection accuracy. Dense planting leads to frequent occlusion of disease targets, increasing detection difficulty. Additionally, disease symptoms in greenhouse environments manifest in diverse forms and are often accompanied by mixed infections (<xref ref-type="bibr" rid="B34">V&#xe1;sconez et&#xa0;al., 2024</xref>). These characteristics make methods that perform well in laboratory environments often struggle to achieve expected results in practical greenhouse applications. The transition from controlled laboratory settings to complex greenhouse environments highlights fundamental challenges that most existing approaches fail to address adequately.</p>
<p>Critical analysis of existing approaches reveals three fundamental research gaps that this work addresses: First, the feature preservation gap - most methods prioritize overall detection accuracy but fail to preserve the fine-grained visual details essential for early-stage disease detection when symptoms are subtle. Second, the scale adaptation gap - current architectures inadequately handle the multi-scale nature of disease manifestations, from macro-level patterns visible to human observers to micro-level changes detectable only through careful feature analysis. Third, the environmental adaptation gap - existing feature fusion strategies lack dynamic mechanisms to handle the varying visual complexity introduced by greenhouse environmental factors such as condensation on covering materials, structural shadows, and plant growth density variations.</p>
<p>However, existing research still has the following limitations: First, most methods are developed for disease images with single backgrounds under laboratory conditions, without fully considering the unique characteristics of greenhouse environments; Second, existing models perform poorly when dealing with complex situations like occlusion and lighting variations in greenhouse environments; Third, the balance between real-time performance and accuracy remains unresolved. As <xref ref-type="bibr" rid="B7">Bouni et&#xa0;al. (2024)</xref> and <xref ref-type="bibr" rid="B1">Abdalla et&#xa0;al. (2024)</xref> point out, developing detection systems adapted to complex greenhouse environments remains a challenging problem requiring urgent solutions.</p>
<p>To address these issues, this study proposes a vegetable disease detection method YOLO-vegetable based on improved YOLOv10 for greenhouse environments. Our experiments are validated on disease image datasets collected from multiple real greenhouse environments. The experimental data includes vegetable disease images under different lighting conditions, planting densities, and growth stages, fully reflecting the characteristics of greenhouse environments. Through comparative experiments with existing mainstream methods, we validate the superiority of our proposed method in greenhouse environments.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Literature review</title>
<p>Deep learning technology has made significant progress in agricultural applications, particularly demonstrating great potential in plant disease detection and recognition. Accurate recognition and early warning of vegetable diseases are crucial for ensuring agricultural production and food security. With the rapid development of computer vision and deep learning technologies, image-based automatic vegetable disease detection methods have gradually become a research hotspot (<xref ref-type="bibr" rid="B27">Paul et&#xa0;al., 2025</xref>). Deep learning methods have shown excellent performance in disease recognition tasks, mainly benefiting from their powerful feature extraction and representation capabilities. Many scholars have conducted in-depth research from different perspectives, proposing various deep learning methods based on Convolutional Neural Networks (CNNs) and object detection networks like the YOLO series (<xref ref-type="bibr" rid="B33">Upadhyay et&#xa0;al., 2025</xref>; <xref ref-type="bibr" rid="B2">Ali et&#xa0;al., 2024</xref>). Currently, research in this field mainly focuses on object detection network design, feature extraction optimization, data augmentation strategies, and multi-modal fusion.</p>
<sec id="s2_1">
<label>2.1</label>
<title>Innovative strategies in object detection network design</title>
<p>In object detection network design, researchers have proposed multiple improvement strategies. With the development of deep learning technology, object detection networks continue to evolve. The Pruned-YOLO v5s+Shuffle model proposed by <xref ref-type="bibr" rid="B38">Xu et&#xa0;al. (2022)</xref> employs channel pruning method, achieving 93.2% detection accuracy in complex backgrounds. The Yolov5-ECA-ASFF network proposed by <xref ref-type="bibr" rid="B43">Zhang et&#xa0;al. (2024)</xref> enhances detection performance by integrating ECA and ASFF modules. <xref ref-type="bibr" rid="B21">Lin et&#xa0;al. (2024)</xref> optimized the YOLO model through combining mixed data augmentation and osprey search strategy, realizing tomato biotic stress detection. The WCG-VMamba model developed by <xref ref-type="bibr" rid="B35">Wang et&#xa0;al. (2024)</xref> introduces wavy vision Mamba network, effectively capturing semantic correlations between image features and text features, further improving detection performance in complex backgrounds. The cross-domain dynamic attention mechanism designed by <xref ref-type="bibr" rid="B26">Mo and Wei (2024)</xref> effectively solves uneven illumination problems. <xref ref-type="bibr" rid="B25">Mhala et&#xa0;al. (2025)</xref> addressed class imbalance issues through model compression and knowledge distillation techniques, achieving efficient model deployment. These studies indicate that object detection network design is evolving towards better adaptation to complex environments and higher accuracy.</p>
<p>Despite promising results in agriculture, existing YOLO-based methods still face fundamental limitations in greenhouse applications: (1) Standard strided convolutions in YOLO backbones sacrifice spatial resolution for computational efficiency, but disease symptoms often manifest as subtle texture changes requiring preservation of fine-grained details; (2) Traditional feature pyramid networks inadequately handle the extreme scale variation of disease symptoms, from macro-level leaf discoloration spanning hundreds of pixels to micro-level lesions occupying fewer than 20 pixels; (3) Fixed feature fusion weights in existing architectures cannot adapt to the dynamic visual complexity of greenhouse environments where lighting conditions, plant density, and background complexity vary significantly. Our ADEConv module specifically addresses the feature preservation challenge while maintaining computational efficiency.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Breakthrough progress in feature extraction optimization</title>
<p>In feature extraction optimization, the introduction of various innovative mechanisms has led to significant breakthroughs. <xref ref-type="bibr" rid="B22">Liu et&#xa0;al. (2021)</xref> proposed region and loss reweighting methods, providing new insights for feature extraction optimization. The EFDet model developed by <xref ref-type="bibr" rid="B23">Liu et&#xa0;al. (2024)</xref> improves detection effects in complex backgrounds by fusing features from different levels. <xref ref-type="bibr" rid="B39">Yan et&#xa0;al. (2024)</xref> proposed an adaptive deep transfer learning framework for mixed subdomains, significantly improving cross-species disease diagnosis performance. Notably, <xref ref-type="bibr" rid="B18">Kang et&#xa0;al. (2024)</xref> proposed a cascade framework combining detector and tracker, significantly reducing computational complexity while maintaining high accuracy, providing a feasible solution for practical application scenarios. <xref ref-type="bibr" rid="B29">Sun et&#xa0;al. (2025)</xref> proposed a new tomato disease recognition method based on the DeiT model, significantly improving detection accuracy in complex environments through improved feature extraction and multi-scale feature fusion mechanisms.</p>
<p>While attention-based approaches show promise, most existing methods apply static attention weights. <xref ref-type="bibr" rid="B9">Chang et&#xa0;al. (2024)</xref> improved wheat disease recognition through DenseNet modifications, but their approach lacks the dynamic adaptability required for greenhouse environmental variations. The AAFS mechanism differs fundamentally from existing attention approaches: Unlike SE-Net which focuses solely on channel attention through global average pooling, AAFS integrates both channel and spatial attention through parallel pathways. Compared to CBAM which applies channel and spatial attention sequentially, our approach enables simultaneous processing and dynamic weight fusion. Unlike ECA-Net&#x2019;s 1D convolution for channel attention, AAFS employs adaptive group convolution with channel shuffling for enhanced feature interaction.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Innovative development in data augmentation strategies</title>
<p>In data augmentation strategies, researchers have proposed a series of innovative methods to address the unique challenges in greenhouse environments. The multi-scale feature enhancement strategy proposed by <xref ref-type="bibr" rid="B31">Tian et&#xa0;al. (2022)</xref> significantly improved the model&#x2019;s recognition ability for disease regions. <xref ref-type="bibr" rid="B19">Karantoumanis et&#xa0;al. (2024)</xref> developed a strategic data augmentation method achieving a 37% accuracy improvement in legume crop disease detection. <xref ref-type="bibr" rid="B43">Zhang et&#xa0;al. (2024)</xref> proposed feature transfer and small target oversampling methods based on CycleGAN, effectively improving sample imbalance issues and successfully achieving precise recognition of early eggplant wilt disease. <xref ref-type="bibr" rid="B17">Johri et&#xa0;al. (2024)</xref> combined deep transfer learning with data augmentation, achieving significant results in small sample scenarios, providing new ideas for solving data insufficiency problems.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Exploration in multi-modal fusion</title>
<p>In multi-modal fusion, researchers have gradually begun to focus on the synergistic use of multi-source information. <xref ref-type="bibr" rid="B40">Yang et&#xa0;al. (2024)</xref> innovatively proposed a language-vision fusion framework, demonstrating excellent performance in tomato disease segmentation tasks. <xref ref-type="bibr" rid="B15">Hu et&#xa0;al. (2024)</xref> achieved deep fusion of spectral information and RGB images, significantly improving disease detection accuracy. <xref ref-type="bibr" rid="B44">Zhao et&#xa0;al. (2025)</xref> successfully implemented complementary fusion of healthy and diseased leaf information using Double Generative Adversarial Networks (DoubleGAN), providing new ideas for disease detection in small sample scenarios.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Small object detection challenges in complex backgrounds</title>
<p>Regarding small target localization and detection recognition in complex backgrounds, the unique characteristics of greenhouse environments bring distinct challenges to disease detection. <xref ref-type="bibr" rid="B5">Barbedo (2019)</xref> research showed that disease recognition faces challenges of small target size, blurred target features, and occlusion problems. <xref ref-type="bibr" rid="B32">Toda and Okura (2019)</xref> revealed the decision mechanism of CNNs in plant disease diagnosis under complex environments. <xref ref-type="bibr" rid="B20">Kumar et&#xa0;al. (2023)</xref> proposed a bidirectional feature attention pyramid network, effectively enhancing the model&#x2019;s detection capability for targets of different scales. <xref ref-type="bibr" rid="B45">Zhou et&#xa0;al. (2023)</xref> innovatively introduced weakly supervised learning into disease feature segmentation, providing new approaches for small target detection. <xref ref-type="bibr" rid="B41">Ye et&#xa0;al. (2024)</xref> proposed an adaptive small target detection framework, significantly improving detection performance in low-light environments by integrating EnlightenGAN networks. <xref ref-type="bibr" rid="B14">Hari and Singh (2025)</xref> proposed an adaptive knowledge transfer method based on federated deep learning, significantly improving model convergence and accuracy through intelligent weight transfer technology optimizing knowledge integration between parent and child entities.</p>
<p>However, existing research still faces severe challenges in complex, unstructured greenhouse environments. First, image acquisition in greenhouse environments suffers from serious quality degradation issues, including image blur, noise interference, and uneven illumination, leading to significant false detections and missed detections in practical applications. Second, vegetable disease symptoms often manifest as small local areas of pathological changes, and these subtle features are easily lost during feature extraction, making them difficult to capture accurately (<xref ref-type="bibr" rid="B28">Qing et&#xa0;al., 2023</xref>). Furthermore, feature expression and multi-scale feature fusion mechanisms under complex background interference remain unresolved (<xref ref-type="bibr" rid="B8">Castillo-Girones et&#xa0;al., 2025</xref>).</p>
<p>To address these issues, this study proposes a high-precision localization and detection algorithm (YOLO-vegetable) for vegetable disease targets in greenhouse environments, based on the computationally efficient YOLOv10n single-stage object detection network. The algorithm contains three core innovative modules: First, we design the Adaptive Detail Enhancement Convolution (ADEConv) module, which significantly improves fine-grained feature retention capability while maintaining computational efficiency through dynamic adjustment of convolution kernel parameters; Second, we construct the Multi-granularity Feature Fusion Detection Layer (MFLayer), which achieves precise localization of small targets through hierarchical integration of feature information at different scales; Finally, we propose the Inter-layer Dynamic Fusion Pyramid Network (IDFNet), combining with Attention-guided Adaptive Feature Selection (AAFS) mechanism, significantly enhancing the model&#x2019;s key information extraction capability by establishing dynamic association weights between feature layers.</p>
</sec>
</sec>
<sec id="s3" sec-type="materials|methods">
<label>3</label>
<title>Materials and methods</title>
<sec id="s3_1">
<label>3.1</label>
<title>YOLO-vegetable model for vegetable disease detection</title>
<p>Based on the characteristics of vegetable disease targets requiring detection, this study proposes a detection algorithm model YOLO-vegetable targeting greenhouse environments. Taking YOLOv10n, which has the smallest parameter count in the detection-performance-excellent YOLOv10 series, as the baseline model, we redesigned the backbone network and neck network of the original model. The structure of YOLO-vegetable is shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>YOLO-vegetable network architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g001.tif">
<alt-text content-type="machine-generated">Flowchart depicting a neural network architecture. The left section, labeled &#x201c;Backbone,&#x201d; includes components like ADEConv, C2f, SPPF, and PSA. The center &#x201c;IDFNet&#x201d; section contains modules such as Concat, AAFS, Upsample, GhostConv, and MF Layer. The right &#x201c;Head&#x201d; section shows output paths with &#x201c;One-to-many&#x201d; and &#x201c;One-to-one&#x201d; heads, each leading to &#x201c;Regression&#x201d; and &#x201c;Classification&#x201d; units, labeled P2, P3, P4, and P5.</alt-text>
</graphic>
</fig>
<p>The diagram illustrates the complete network structure with backbone (left), neck network with IDFNet (center), and detection heads (right). Red boxes highlight our proposed modules: ADEConv modules replace traditional strided convolutions in the backbone, MFLayer provides multi-granularity feature fusion for small target detection, and AAFS mechanisms enable adaptive feature selection throughout the neck network. Input images (640&#xd7;640) flow through the backbone for feature extraction, then through IDFNet for multi-scale feature fusion, finally reaching dual detection heads for classification and regression outputs.</p>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Design of ADEConv</title>
<p>Convolutional Neural Networks (CNNs) are widely applied in computer vision tasks. In YOLOv10 algorithm, CNN is a core part of its architecture. In traditional CNN design, strided convolution is typically used for downsampling operations to extract spatial features, with common convolution kernel sizes of 3&#xd7;3 or larger. Strided convolution achieves downsampling by setting stride greater than 1 during convolution operations, meaning the convolution kernel moves multiple pixels at a time rather than pixel by pixel. For example, when stride is set to 2, the convolution kernel moves 2 pixels each time, where only one out of every two pixels in the input feature map is covered by the convolution kernel, thus halving the output feature map dimensions. Because strided convolution skips some input data, important local information may not be captured. Although this feature downsampling can aggregate contextual information and achieve dimension reduction, it comes at the cost of losing detail information, challenging the model&#x2019;s ability to recognize and learn small target features.</p>
<p>Traditional pooling layers can also reduce feature map resolution and computational cost, but during this process, information about small objects may be excessively compressed or completely lost, leading to decreased detection performance. Therefore, when using these modules for downsampling, details of vegetable disease targets are inevitably lost, affecting the network&#x2019;s ability to extract fine details of small disease spots. Moreover, diseased areas in vegetable images occupy extremely small areas, and uneven lighting in greenhouse environments makes it necessary for the network to extract more detailed information to improve small target recognition ability.</p>
<p>To address the aforementioned issues, this study replaces the traditional strided convolution modules in YOLOv10&#x2019;s backbone network with ADEConv modules, which improve small object detection performance by preserving fine-grained information and avoiding excessive compression of image features. The replacement process is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Backbone network architecture comparison. Left: Original YOLOv10n backbone using standard strided convolutions (Conv) and SCDown modules. Right: Our improved backbone with ADEConv modules replacing all downsampling operations. The ADEConv modules preserve fine-grained features while achieving the same spatial dimension reduction, addressing the information loss problem inherent in traditional strided convolutions. Each P1-P5 represents feature maps at different scales (1/2, 1/4, 1/8, 1/16, 1/32 of input resolution respectively).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g002.tif">
<alt-text content-type="machine-generated">Diagram comparing two convolutional neural network architectures. The left side shows a traditional series: Conv layers labeled P1 to P5 with an SCDown at the end. The right side replaces Conv with ADEConv layers, following the same P1 to P5 layout. An arrow separates the two models, indicating a transformation from Conv to ADEConv implementation.</alt-text>
</graphic>
</fig>
<p>The ADEConv module primarily consists of a Space-to-depth Module and a Non-strided Ghost Convolution Block (<xref ref-type="bibr" rid="B13">Han et&#xa0;al., 2020</xref>), replacing all strided convolution blocks in YOLOv10&#x2019;s backbone network. The Space-to-depth Module first performs pixel-wise division and rearranges pixels from each block into depth channels, achieving spatial compression of the input feature map. This reorganization not only halves the feature map&#x2019;s spatial dimensions but also preserves all original information of the processed pixels, effectively avoiding potential detail loss that might occur during traditional strided convolution&#x2019;s spatial compression process. The module&#x2019;s main structure is shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>ADEConv module architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g003.tif">
<alt-text content-type="machine-generated">Diagram depicting a neural network architecture divided into two sections: a Space-to-depth Module and a Non-strided Ghost Convolution Block. On the left, a cube labeled \(X_{in}\) undergoes slicing, resulting in smaller cubes which combine into a larger cube \(X_{spd}\). On the right, another cube \(X_{in}\) is convolved where stride equals one, generating two smaller cubes that combine into a final output cube \(X_{out}\).</alt-text>
</graphic>
</fig>
<p>Here, <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the ADEConv module&#x2019;s input feature map, S represents the spatial dimension width/height value of the input feature map, <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the input feature map&#x2019;s channel number, <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the Space-to-depth Module&#x2019;s output feature map, <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the output feature map&#x2019;s channel number, and <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the ADEConv module&#x2019;s output feature map.</p>
<p>The first operation of the Space-to-depth Module is feature map slicing, with its formula being:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where X denotes the input feature map, h and w are the starting indices for feature map height and width respectively, S is the input feature map dimension, and scale is the slicing stride. When scale=2, extracting values every 2 elements yields the following four feature maps:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Finally, channel concatenation is performed:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where Concat[] represents the Channel-wise Concatenation operation. While preserving detail information, the Space-to-depth Module reduces the feature map&#x2019;s spatial dimensions. Subsequently, the Non-strided Ghost Convolution Block reduces channel numbers, with its formula being:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mn>5</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents group convolution operation with a 5&#xd7;5 kernel size, <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents a 1&#xd7;1 convolution transformation function using channel size of <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo stretchy="false">/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Through these operations, the ADEConv module can achieve downsampling operations while maximally preserving all detail information from the original image without significantly increasing computational cost.</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Design of MFLayer</title>
<p>In traditional YOLO series network design, the Path Aggregation Feature Pyramid Network (PAFPN) adopts a structure of downsampling followed by upsampling then downsampling again, combined with skip connections to enhance information exchange between feature maps. Feature maps are divided into five levels from P1 to P5 based on their spatial reduction ratio relative to the input image (1/2, 1/4, 1/8, 1/16, 1/32).</p>
<p>After multiple downsampling operations, some low-level features may gradually be lost. Although skip connections between feature maps of the same level during downsampling and upsampling help recover detail information lost due to consecutive convolutions and pooling operations, for extremely small targets, the original structure&#x2019;s restoration of details remains insufficient after five downsampling operations followed by only two upsampling operations, affecting network detection performance.</p>
<p>To address this issue, we introduce the MFLayer in the neck network to preserve extremely small target detail features. We fuse the P2 feature layer with downsampling factor of two from the backbone network and the P2 feature layer obtained after three upsampling operations from the P5 feature layer, and directly use it as input for the small target detection head. This design aims to enhance the model&#x2019;s localization and recognition capability for extremely small-sized objects by preserving sufficient low-level features. The principle of the MFLayer is shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>MFLayer schematic diagram. <bold>(a)</bold> Original Network: Traditional YOLO architecture processes features through standard downsampling and upsampling paths, with P2-P5 representing feature pyramid levels at different scales. <bold>(b)</bold> With MFLayer: Our enhanced architecture introduces additional connections (red arrows) that preserve high-resolution P2 features and directly integrate them with upsampled deep features, enabling better small target detection through multi-granularity feature fusion.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g004.tif">
<alt-text content-type="machine-generated">Diagram showing two network architectures. (a) Original Network: Five stacked layers marked P2 to P5 with arrows indicating connections. (b) With MFLayer: Similar stack but includes additional red arrows indicating extra connections between nodes P3, P4, and P5.</alt-text>
</graphic>
</fig>
<p>As shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, (a) Original Network: Traditional YOLO architecture processes features through standard downsampling and upsampling paths, with P2-P5 representing feature pyramid levels. (b) With MFLayer: Our enhanced architecture introduces additional connections (red arrows) that preserve high-resolution P2 features and directly integrate them with upsampled deep features. This strategy of combining low-level detail features with high-level semantic features not only helps improve detection effects for small targets but can also maintain the model&#x2019;s computational efficiency to some extent. Through this approach, the model can more accurately capture and identify small objects in images. The MFLayer design offers significant advantages over traditional feature fusion approaches by establishing direct connections between high-resolution and low-resolution feature maps. This capability directly addresses one of the most significant challenges in greenhouse disease detection, where early-stage symptoms often manifest as subtle lesions easily lost during conventional feature downsampling.</p>
</sec>
<sec id="s3_1_3">
<label>3.1.3</label>
<title>Design of IDFNet</title>
<p>Traditional YOLO series networks use PAFPN as their neck network structure, where there is no information exchange between each feature map layer and the backbone, potentially leading to loss of some detail features. The preservation of detail features is crucial for small target recognition. The Bi-directional Feature Pyramid Network (BiFPN) adds cross-scale fusion layers compared to PAFPN (<xref ref-type="bibr" rid="B30">Tan et&#xa0;al., 2020</xref>), achieving feature flow from top-down and bottom-up, and optimizing the feature fusion process by adding weights to each feature input, which helps preserve more useful information. Since BiFPN introduces dynamic weights, these weights are optimized through backpropagation during network training, which might increase the network&#x2019;s computational burden and potentially lead to training instability in early stages due to uncertainty in initial weight values.</p>
<p>To address these issues, this study redesigns the original neck network, proposing the IDFNet. This network introduces a feature propagation path from backbone to downsampling path to reduce the loss of small target features during propagation. By introducing cross-layer feature propagation paths, we establish direct connections between the backbone network and feature pyramid network, significantly reducing information loss of fine-grained features during multiple downsampling processes. The output layers from the feature extraction network are fed into P3 layer (low-level features), P4 layer (mid-level features), and P5 layer (high-level features), and BiFPN fusion method is repeated 3 times between P3, P4, and P5 layers, implementing multi-scale feature fusion. Each fusion can extract higher-level, more abstract features based on existing foundations, improving detection accuracy. The overall architecture of IDFNet is shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Comparison of feature pyramid network architectures. <bold>(a)</bold> FPN: Basic top-down feature fusion with unidirectional information flow from high-level to low-level features. <bold>(b)</bold> PAFPN: Bidirectional feature fusion with additional bottom-up pathway enabling information exchange between different pyramid levels. <bold>(c)</bold> BiFPN: Weighted bidirectional fusion with cross-scale connections and learnable fusion weights. <bold>(d)</bold> IDFNet (Ours): Enhanced architecture with additional backbone-to-neck connections (green arrows) and dynamic fusion weights through AAFS mechanism for improved feature propagation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g005.tif">
<alt-text content-type="machine-generated">Diagram illustrating a progression of interconnected processes labeled P2 to P5, in four panels (a to d). Panel (a) shows a linear sequence. Panel (b) introduces feedback loops. Panel (c) adds additional interconnected pathways. Panel (d) depicts a complex network with multiple feedback loops and connections, suggesting an evolution of dependency or interaction.</alt-text>
</graphic>
</fig>
<p>As shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, (a) FPN: Basic top-down feature fusion. (b) PAFPN: Bidirectional feature fusion with additional bottom-up pathway. (c) BiFPN: Weighted bidirectional fusion with cross-scale connections. (d) IDFNet (Ours): Enhanced architecture with additional backbone-to-neck connections (green arrows) and dynamic fusion weights.</p>
<p>Simultaneously, we design the AAFS mechanism as the core feature fusion strategy. Unlike traditional BiFPN using fixed weight allocation methods, the AAFS mechanism dynamically calculates fusion weights by comprehensively analyzing channel-dimension and spatial-dimension correlations of feature maps, enabling the network to adaptively enhance features crucial for detection tasks. This strategy based on feature correlation adaptive selection not only improves the model&#x2019;s detection sensitivity to subtle disease features but also enhances feature expression&#x2019;s discriminative ability across different scales. The principle of AAFS is shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>AAFS module architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g006.tif">
<alt-text content-type="machine-generated">Diagram of a neural network layer showing input passing through average and max pooling, followed by two one-by-one convolutions and ReLU activation. Features are then concatenated, passed through a channel shuffle and group convolution, resulting in the output.</alt-text>
</graphic>
</fig>
<p>Let X be the feature map input, obtaining its channel-dimension and spatial-dimension features:</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mfrac bevelled="true">
<mml:mi>C</mml:mi>
<mml:mi>r</mml:mi>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>C</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents channel features, <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents spatial features, <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents 1&#xd7;1 convolution transformation function with C channels, r is the channel reduction ratio, max(0,&#xb7;) represents the ReLU activation function, <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>C</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents global average pooling operation across spatial dimensions, <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:msubsup>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represent global average pooling and global max pooling operations across channel dimensions respectively.</p>
<p>Subsequently, the features from both dimensions are added and concatenated with input X, followed by channel shuffling operation, then passing through group convolution and Sigmoid operation to obtain fusion weight W:</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where &#x3c3; represents Sigmoid operation, <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents group convolution operation with 5&#xd7;5 kernel size, CS() represents channel shuffle operation.</p>
</sec>
<sec id="s3_1_4">
<label>3.1.4</label>
<title>Variable definitions</title>
<p>
<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> summarizes key variables, their symbols, definitions, and numerical values/ranges used in the study.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Variable definitions.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Symbol</th>
<th valign="middle" align="left">Definition</th>
<th valign="middle" align="left">Value/Range</th>
<th valign="middle" align="left">Source</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">
<italic>C</italic>1&#x200b;</td>
<td valign="middle" align="left">Input channels of ADEConv</td>
<td valign="middle" align="left">
<italic>C</italic>1&#x200b;=64</td>
<td valign="middle" align="left">Backbone network configuration</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>C</italic>2&#x200b;</td>
<td valign="middle" align="left">Output channels of ADEConv</td>
<td valign="middle" align="left">
<italic>C</italic>2&#x200b;=128</td>
<td valign="middle" align="left">
<xref ref-type="disp-formula" rid="eq4">Equation 4</xref>
</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>H</italic>,<italic>W</italic>
</td>
<td valign="middle" align="left">Height/Width of input feature maps</td>
<td valign="middle" align="left">
<italic>H</italic>=<italic>W</italic>=640</td>
<td valign="middle" align="left">Image resolution setting</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>&#x3c3;</italic>
</td>
<td valign="middle" align="left">Gaussian noise intensity</td>
<td valign="middle" align="left">&#x3c3;&#x2208;[0.1,0.3]<break/>
<italic>&#x3c3;</italic>&#x2208;[0.1,0.3]</td>
<td valign="middle" align="left">Noise robustness experiments</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>&#x3b1;</italic>
</td>
<td valign="middle" align="left">Learning rate decay factor</td>
<td valign="middle" align="left">
<italic>&#x3b1;</italic>=0.95</td>
<td valign="middle" align="left">Training hyperparameters</td>
</tr>
<tr>
<td valign="middle" align="left">
<italic>r</italic>
</td>
<td valign="middle" align="left">Channel reduction ratio in AAFS</td>
<td valign="middle" align="left">
<italic>r</italic>=4</td>
<td valign="middle" align="left">
<xref ref-type="disp-formula" rid="eq5">Equation 5</xref>
</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Vegetable disease image dataset</title>
<p>To ensure the dataset encompasses diverse greenhouse environments and meets the model&#x2019;s requirements for handling complex backgrounds, occlusion, blurred disease features, and small target detection, this study employs our self-built Vegetable Disease Dataset (VDD), comprising 15,000 images involving 3 major facility vegetables (tomato, cucumber, pepper) and their 15 common diseases along with healthy samples (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>; <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). Data collection was conducted in controlled greenhouse facilities with temperature at 22-28&#xb0;C (day) and 18-22&#xb0;C (night), and relative humidity at 60-75%. Images were captured using professional high-resolution cameras at 30-50cm distance across four growth stages (seedling, vegetative, flowering, fruiting) under diverse weather conditions to ensure dataset robustness. The dataset is annotated following YOLO format specifications. Dataset annotation was performed by certified plant pathologists following standardized protocols. Each disease instance was annotated with precise bounding boxes. The dataset is divided into training set, validation set, and test set in a 7:2:1 ratio. This dataset contains vegetable disease targets under various weather conditions.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Selected samples of vegetable disease images.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g007.tif">
<alt-text content-type="machine-generated">Grid of plant leaves showing various health and disease conditions. Top row: healthy tomato, tomato gray mold, gray leaf spot, black spot, late blight. Middle row: healthy cucumber, target spot, powdery mildew, angular spot, downy mildew. Bottom row: healthy pepper, leaf spot, powdery mildew, black spot, early blight.</alt-text>
</graphic>
</fig>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Sample counts of vegetable disease types.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">No.</th>
<th valign="middle" rowspan="2" align="center">Disease type</th>
<th valign="top" colspan="3" align="center">Number of images</th>
</tr>
<tr>
<th valign="top" align="center">Training set</th>
<th valign="top" align="center">Validation set</th>
<th valign="top" align="center">Test set</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">A1</td>
<td valign="top" align="center">Tomato health</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">A2</td>
<td valign="top" align="center">Tomato gray mold</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">A3</td>
<td valign="top" align="center">Tomato gray leaf spot</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">A4</td>
<td valign="top" align="center">Tomato black spot</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">A5</td>
<td valign="top" align="center">Tomato late blight</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">B1</td>
<td valign="top" align="center">Cucumber health</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">B2</td>
<td valign="top" align="center">Cucumber target spot</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">B3</td>
<td valign="top" align="center">Cucumber powdery mildew</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">B4</td>
<td valign="top" align="center">Cucumber angular spot</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">B5</td>
<td valign="top" align="center">Cucumber downy mildew</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">C1</td>
<td valign="top" align="center">Pepper health</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">C2</td>
<td valign="top" align="center">Pepper leaf spot</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">C3</td>
<td valign="top" align="center">Pepper powdery mildew</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">C4</td>
<td valign="top" align="center">Pepper black spot</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">C5</td>
<td valign="top" align="center">Pepper early blight</td>
<td valign="middle" align="center">700</td>
<td valign="middle" align="center">200</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">Total</td>
<td valign="middle" align="center">
</td>
<td valign="middle" align="center">10500</td>
<td valign="middle" align="center">3000</td>
<td valign="middle" align="center">1500</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> showcases representative samples from our dataset, illustrating the diversity of disease manifestations across different vegetable types and growth stages. The images demonstrate varying symptom presentations, from early-stage subtle discolorations to advanced necrotic lesions, captured under diverse lighting conditions and viewing angles. This diversity ensures models trained on our dataset develop robust generalization capabilities applicable to real-world greenhouse scenarios.</p>
</sec>
</sec>
<sec id="s4" sec-type="results">
<label>4</label>
<title>Results and discussion</title>
<sec id="s4_1">
<label>4.1</label>
<title>Experimental environment and parameter configuration</title>
<p>The experimental platform uses Ubuntu22.04 as the operating system, equipped with Intel(R) Xeon(R) Gold 5418Y processor with a main frequency of 2.00 GHz. The system memory is 32GB, with an Nvidia GeForce RTX 4090 graphics card having 24GB memory capacity. The PyTorch framework version is 2.2.2+cu121, and Python version is 3.10.0. Input image resolution is uniformly set to 640&#xd7;640 to ensure the clarity of targets at different scales in feature maps, adapting to the model&#x2019;s requirements for small target detection. During training, the model&#x2019;s initial learning rate is set to 0.01, batch size to 16, momentum to 0.937, weight decay coefficient to 0.0005, and training epochs to 100. To further enhance the model&#x2019;s robustness, all experiments are conducted without any form of pre-trained weights, and all experiments use consistent hyperparameters for training and validation to ensure comparability of experimental results.</p>
<p>Data augmentation techniques, including mosaic and mixup, were applied to enhance dataset diversity. Our augmentation pipeline also included random rotation (&#xb1; 15&#xb0;), horizontal and vertical flipping, and adjustments to brightness (&#xb1; 25%), contrast (&#xb1; 20%), and saturation (&#xb1; 15%) to simulate the variable lighting conditions in greenhouse environments. To address class imbalance issues, we employed oversampling for minority disease classes, ensuring balanced representation during training while maintaining authentic image characteristics.</p>
<p>For hyperparameter optimization, we conducted a systematic grid search to identify optimal values. The learning rate was initialized at 0.01 and adjusted using a cosine annealing scheduler with warm restarts. Weight decay was set to 0.0005, and momentum maintained at 0.937 throughout training. These parameters were selected after evaluating 16 different configurations, with the final values providing the best balance between convergence speed and model generalization.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Evaluation metrics</title>
<p>To comprehensively evaluate YOLO-vegetable model&#x2019;s balanced performance in terms of speed and accuracy, this study selects Precision, Recall, Average Precision (AP), Mean Average Precision (mAP), Parameters, FLOPs, and Inference Time as evaluation metrics. In object detection tasks, mAP@0.5 and mAP@0.5:0.95 serve as primary evaluation metrics, capable of comprehensively evaluating model performance. Specifically, mAP@0.5 represents average precision at Intersection over Union (IoU) threshold of 0.5; mAP@0.5:0.95 reflects model stability under different IoU thresholds. Meanwhile, through evaluating parameter count and computational complexity, we provide important references for practical deployment.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Experimental process</title>
<p>To significantly improve the accuracy and efficiency of vegetable disease target detection, we propose YOLO-vegetable. The experimental process includes three key phases, as shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>The flow chart of vegetable disease detection.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g008.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a process with three stages: Data Preparation, Model Construction, and Disease Detection. Data Preparation includes steps from data collection to partitioning. Model Construction involves baseline model creation, improvement, training, and validation. Disease Detection includes input of disease images, model testing, output classification, and performance evaluation. Arrows indicate the flow between stages.</alt-text>
</graphic>
</fig>
<p>To validate the effectiveness of the proposed model, experiments were conducted on our self-built vegetable disease dataset. The training and validation curves of the proposed model&#x2019;s box loss, dfl loss, classification loss, and other performance metrics including precision, recall, mAP@0.5, and mAP@0.5:0.95 are shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, with iteration count on the horizontal axis.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Performance metrics of the proposed YOLO-vegetable model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g009.tif">
<alt-text content-type="machine-generated">Ten line graphs showing training and validation metrics over epochs. The top row displays &#x201c;train/box_loss,&#x201d; &#x201c;train/obj_loss,&#x201d; &#x201c;train/cls_loss,&#x201d; &#x201c;metrics/precision,&#x201d; and &#x201c;metrics/recall,&#x201d; all showing downward trends and fluctuations. The bottom row includes &#x201c;val/box_loss,&#x201d; &#x201c;val/obj_loss,&#x201d; &#x201c;val/cls_loss,&#x201d; &#x201c;metrics/mAP_0.5,&#x201d; and &#x201c;metrics/mAP_0.5:0.95,&#x201d; with similar patterns, indicating performance improvements over time. Each graph has &#x201c;Epoch&#x201d; on the x-axis and &#x201c;Value&#x201d; on the y-axis.</alt-text>
</graphic>
</fig>
<p>As shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, during the 100 training iterations, the loss exhibit stable convergence patterns, gradually stabilizing as training progresses. Similarly, the validation loss demonstrate consistent convergence behavior, reaching steady states by the final epochs. Observing the model&#x2019;s mAP@0.5 and mAP@0.5:0.95 convergence curves, performance metric curves begin to stabilize after 50 iterations. Finally, the model achieves excellent performance on the test set: mAP@0.5 reaches around 95%, and mAP@0.5:0.95 reaches approximately 60%. Meanwhile, the model demonstrates good precision and recall performance, indicating strong generalization ability and stability in vegetable detection tasks. The YOLO-vegetable model achieves a parameter count of 3.8M and a computational complexity of 14.7 GFLOPs, making it highly efficient for real-time deployment in resource-constrained environments.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Experimental results</title>
<p>To comprehensively evaluate YOLO-vegetable model&#x2019;s performance in detecting various vegetable diseases, testing was conducted based on our self-built dataset. <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> shows the detection results of YOLO-vegetable model for these different types of diseases.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Detection results for different disease types.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">No.</th>
<th valign="middle" align="center">Disease type</th>
<th valign="middle" align="center">Precision(%)</th>
<th valign="middle" align="center">Recall(%)</th>
<th valign="middle" align="center">AP50(%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">A1</td>
<td valign="top" align="center">Tomato health</td>
<td valign="middle" align="center">96.8</td>
<td valign="middle" align="center">95.4</td>
<td valign="middle" align="center">96.2</td>
</tr>
<tr>
<td valign="top" align="center">A2</td>
<td valign="top" align="center">Tomato gray mold</td>
<td valign="middle" align="center">95.2</td>
<td valign="middle" align="center">94.8</td>
<td valign="middle" align="center">95.0</td>
</tr>
<tr>
<td valign="top" align="center">A3</td>
<td valign="top" align="center">Tomato gray leaf spot</td>
<td valign="middle" align="center">94.6</td>
<td valign="middle" align="center">93.9</td>
<td valign="middle" align="center">94.3</td>
</tr>
<tr>
<td valign="top" align="center">A4</td>
<td valign="top" align="center">Tomato black spot</td>
<td valign="middle" align="center">95.8</td>
<td valign="middle" align="center">94.7</td>
<td valign="middle" align="center">95.3</td>
</tr>
<tr>
<td valign="top" align="center">A5</td>
<td valign="top" align="center">Tomato late blight</td>
<td valign="middle" align="center">96.2</td>
<td valign="middle" align="center">95.1</td>
<td valign="middle" align="center">95.7</td>
</tr>
<tr>
<td valign="top" align="center">B1</td>
<td valign="top" align="center">Cucumber health</td>
<td valign="middle" align="center">97.1</td>
<td valign="middle" align="center">96.3</td>
<td valign="middle" align="center">96.8</td>
</tr>
<tr>
<td valign="top" align="center">B2</td>
<td valign="top" align="center">Cucumber target spot</td>
<td valign="middle" align="center">95.4</td>
<td valign="middle" align="center">94.6</td>
<td valign="middle" align="center">95.1</td>
</tr>
<tr>
<td valign="top" align="center">B3</td>
<td valign="top" align="center">Cucumber powdery mildew</td>
<td valign="middle" align="center">94.8</td>
<td valign="middle" align="center">93.9</td>
<td valign="middle" align="center">94.4</td>
</tr>
<tr>
<td valign="top" align="center">B4</td>
<td valign="top" align="center">Cucumber angular spot</td>
<td valign="middle" align="center">95.6</td>
<td valign="middle" align="center">94.8</td>
<td valign="middle" align="center">95.2</td>
</tr>
<tr>
<td valign="top" align="center">B5</td>
<td valign="top" align="center">Cucumber downy mildew</td>
<td valign="middle" align="center">96.4</td>
<td valign="middle" align="center">95.2</td>
<td valign="middle" align="center">95.9</td>
</tr>
<tr>
<td valign="top" align="center">C1</td>
<td valign="top" align="center">Pepper health</td>
<td valign="middle" align="center">97.3</td>
<td valign="middle" align="center">96.5</td>
<td valign="middle" align="center">97.0</td>
</tr>
<tr>
<td valign="top" align="center">C2</td>
<td valign="top" align="center">Pepper leaf spot</td>
<td valign="middle" align="center">95.7</td>
<td valign="middle" align="center">94.9</td>
<td valign="middle" align="center">95.3</td>
</tr>
<tr>
<td valign="top" align="center">C3</td>
<td valign="top" align="center">Pepper powdery mildew</td>
<td valign="middle" align="center">94.9</td>
<td valign="middle" align="center">94.2</td>
<td valign="middle" align="center">94.6</td>
</tr>
<tr>
<td valign="top" align="center">C4</td>
<td valign="top" align="center">Pepper black spot</td>
<td valign="middle" align="center">95.8</td>
<td valign="middle" align="center">94.7</td>
<td valign="middle" align="center">95.3</td>
</tr>
<tr>
<td valign="top" align="center">C5</td>
<td valign="top" align="center">Pepper early blight</td>
<td valign="middle" align="center">96.1</td>
<td valign="middle" align="center">95.3</td>
<td valign="middle" align="center">95.8</td>
</tr>
<tr>
<td valign="middle" align="center"/>
<td valign="middle" align="center">Average</td>
<td valign="middle" align="center">95.8</td>
<td valign="middle" align="center">94.9</td>
<td valign="middle" align="center">95.6</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, YOLO-vegetable model achieves Precision, Recall, and AP values above 90% for 15 vegetable diseases and healthy samples, demonstrating high precision and recall rates. The model&#x2019;s mAP reaches 95.6%, fully proving its excellent performance in handling different types of vegetable diseases. Additionally, the model&#x2019;s outstanding performance in healthy sample recognition helps reduce misdiagnosis and unnecessary treatments.</p>
<p>
<xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref> presents the confusion matrix of our proposed YOLO-vegetable model, showing the proportion of detection results for each category. The horizontal axis represents predicted class numbers, while the vertical axis represents annotated class numbers. In the matrix, squares where predicted classes match annotated classes represent correct algorithm predictions, while other squares represent class confusion cases. From the prediction results, the model demonstrates high detection accuracy with minimal confusion overall, showing only a small proportion of class confusion cases.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Confusion matrix of the proposed YOLO-vegetable model.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g010.tif">
<alt-text content-type="machine-generated">Confusion matrix heatmap showing predicted versus actual classes from A1 to C5. High accuracy is reflected with darker shades, peaking at values like 0.968 for A1 vs A1, 0.971 for B1 vs B1, and 0.973 for C1 vs C1. The diagonal dominance indicates strong model performance.</alt-text>
</graphic>
</fig>
<p>To comprehensively evaluate the detection performance of the proposed YOLO-vegetable model compared to the baseline model, we analyzed the Precision-Recall (PR) curves, which illustrate the trade-off between precision and recall at different confidence thresholds. <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> presents the PR curves for both models.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>PR curves of the proposed YOLO-vegetable model compared to the baseline model. <bold>(a)</bold> YOLO-vegetable: Precision-Recall curve showing superior performance with AP of 0.956, maintaining higher precision values across a wider range of recall thresholds. <bold>(b)</bold> Baseline: YOLOv10n baseline model PR curve with AP of 0.892, demonstrating lower overall detection performance compared to our proposed method.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g011.tif">
<alt-text content-type="machine-generated">Two precision-recall curves compare models for performance analysis. Panel (a) features a red line for &#x201c;YOLO-vegetable,&#x201d; showing slightly higher performance. Panel (b) displays a blue line for &#x201c;Baseline.&#x201d; Both curves show precision decreasing as recall increases.</alt-text>
</graphic>
</fig>
<p>The PR curve analysis reveals that the proposed YOLO-vegetable model (<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11a</bold>
</xref>) achieves superior performance with an Average Precision (AP) of 0.956, representing a significant improvement over the baseline model&#x2019;s AP of 0.892 (<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11b</bold>
</xref>). The YOLO-vegetable model maintains higher precision values across a wider range of recall values, indicating its ability to identify disease instances correctly while minimizing false positives. The enhanced performance demonstrated by the PR curves further validates the effectiveness of our architectural improvements&#x2014;specifically the ADEConv module for preserving fine-grained features, the MFLayer for accurate small target localization, and the IDFNet for enhanced feature fusion. These components collectively contribute to the model&#x2019;s ability to maintain high precision even at high recall thresholds, making it well-suited for real-world greenhouse disease detection scenarios with varying lighting conditions and complex backgrounds.</p>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Ablation study</title>
<p>To systematically evaluate the performance contribution of each core module in the YOLO-vegetable algorithm, this study uses YOLOv10n as the baseline model, progressively introducing the ADEConv, MFLayer, and IDFNet modules. Through comprehensive analysis of model accuracy, computational complexity, and inference time, we validate the optimization effect of each module. Detailed experimental results are shown in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Ablation study results.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Group</th>
<th valign="top" align="center">ADEConv</th>
<th valign="top" align="center">MFLayer</th>
<th valign="top" align="center">IDFNet</th>
<th valign="top" align="center">mAP (%)</th>
<th valign="top" align="left">Parameters (M)</th>
<th valign="top" align="left">FLOPs (G)</th>
<th valign="top" align="left">Time (ms)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">1</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">No</td>
<td valign="middle" align="center">89.2</td>
<td valign="middle" align="center">2.2</td>
<td valign="middle" align="center">6.5</td>
<td valign="middle" align="center">15.6</td>
</tr>
<tr>
<td valign="middle" align="center">2</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">No</td>
<td valign="middle" align="center">94.3</td>
<td valign="middle" align="center">3.6</td>
<td valign="middle" align="center">9.5</td>
<td valign="middle" align="center">16.1</td>
</tr>
<tr>
<td valign="middle" align="center">3</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">No</td>
<td valign="middle" align="center">93.2</td>
<td valign="middle" align="center">2.8</td>
<td valign="middle" align="center">15.1</td>
<td valign="middle" align="center">18.2</td>
</tr>
<tr>
<td valign="middle" align="center">4</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">Yes</td>
<td valign="middle" align="center">94.3</td>
<td valign="middle" align="center">2.7</td>
<td valign="middle" align="center">8.3</td>
<td valign="middle" align="center">15.9</td>
</tr>
<tr>
<td valign="middle" align="center">5</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">No</td>
<td valign="middle" align="center">94.5</td>
<td valign="middle" align="center">3.8</td>
<td valign="middle" align="center">16.7</td>
<td valign="middle" align="center">20.1</td>
</tr>
<tr>
<td valign="middle" align="center">6</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">Yes</td>
<td valign="middle" align="center">94.2</td>
<td valign="middle" align="center">4.2</td>
<td valign="middle" align="center">10.3</td>
<td valign="middle" align="center">17.6</td>
</tr>
<tr>
<td valign="middle" align="center">7</td>
<td valign="top" align="center">No</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">Yes</td>
<td valign="middle" align="center">94.0</td>
<td valign="middle" align="center">3.4</td>
<td valign="middle" align="center">12.9</td>
<td valign="middle" align="center">17.9</td>
</tr>
<tr>
<td valign="middle" align="center">8</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">Yes</td>
<td valign="top" align="center">Yes</td>
<td valign="middle" align="center">95.6</td>
<td valign="middle" align="center">3.8</td>
<td valign="middle" align="center">14.7</td>
<td valign="middle" align="center">18.6</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The experimental results show that introducing the ADEConv module improves mAP@0.5 from 89.2% to 94.3%, significantly enhancing the network&#x2019;s fine-grained feature extraction capability. Although parameter count increases from 2.2M to 3.6M and computational cost increases to 9.5 GFLOPs, it only brings a 0.5ms inference time delay (15.6ms to 16.1ms). While the MFLayer module leads to computational cost increasing to 15.1 GFLOPs with a 2.6ms inference time increase, it performs excellently in maintaining small target detail features, achieving 93.2% mAP@0.5 with only 2.8M parameters. The introduction of IDFNet demonstrates superior feature fusion effects, achieving 94.3% mAP@0.5 with just 2.7M parameters, while maintaining comparable computational cost (8.3 GFLOPs) and inference time (15.9ms).</p>
<p>Further research reveals that the combination of ADEConv and MFLayer achieves 94.5% mAP@0.5, surpassing single-module applications. Although computational cost increases to 16.7 GFLOPs, through reasonable parameter configuration (3.8M), the inference time increase (20.1ms) remains acceptable. This result demonstrates the synergistic effect between detail feature extraction and feature preservation. Building upon this foundation, introducing IDFNet to form the complete YOLO-vegetable model not only further improves mAP@0.5 to 95.6% but also achieves optimization in computational resource utilization: maintaining parameter count at 3.8M, reducing computational cost to 14.7 GFLOPs, and controlling inference time to 18.6ms. This balance between performance improvement and computational overhead fully validates the necessity of innovative modules and their excellent synergistic effects.</p>
<p>To more intuitively demonstrate the performance improvement effects of different modules on the model, <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref> illustrates the trends of model performance as different modules are introduced. From the overall trends in <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>, the progressive introduction of the three innovative modules shows steady performance improvement, with balanced enhancement across all metrics, demonstrating no significant degradation in any indicator while others improve. This balanced performance improvement validates that our proposed improvement strategies are not only necessary but can work collaboratively and mutually reinforce each other, achieving overall optimization of model performance.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Performance contribution comparison of different modules.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g012.tif">
<alt-text content-type="machine-generated">Bar chart showing performance metrics for Baseline, +ADEConv, +MFLayer, and +IDFNet. Measured in mAP@0.5 (blue), Precision (green), and Recall (orange). Performance improves progressively across the models, with +IDFNet showing the highest values.</alt-text>
</graphic>
</fig>
<p>To better understand the model&#x2019;s decision-making process, we visualized the feature activation maps using Grad-CAM techniques (<xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref>). These visualizations demonstrate that our YOLO-vegetable model correctly focuses on disease-affected regions while effectively filtering out background noise. For smaller lesions, the model exhibits precise localization, confirming the effectiveness of our detail-preserving modules. Comparative analysis of activation maps between the baseline model and YOLO-vegetable reveals distinct differences in feature focus. While the baseline model tends to activate broadly across leaf surfaces with disease-like coloration patterns, our model demonstrates more precise localization specifically on the actual disease lesions. This is particularly evident in the second row of <xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref>, where the baseline model shows diffuse activation across multiple spots, while YOLO-vegetable concentrates activation intensity precisely on the primary disease lesions. This improved focus significantly reduces false positives in complex backgrounds with similar color patterns to diseases but different textural features, a common challenge in greenhouse environments with varying light conditions creating shadowing effects that resemble disease symptoms.</p>
<fig id="f13" position="float">
<label>Figure&#xa0;13</label>
<caption>
<p>Feature activation maps using Grad-CAM technique.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g013.tif">
<alt-text content-type="machine-generated">Three sets of images display plant health assessments. Each set contains an original photo, a baseline heatmap, and a YOLO-vegetable model heatmap. The first row shows a tomato with blemishes, the second a leaf with spots, and the third a leaf with curling. Red boxes highlight areas of interest in the heatmaps.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Comparative experiments</title>
<p>
<xref ref-type="fig" rid="f14">
<bold>Figure&#xa0;14</bold>
</xref> presents the comparative experimental results between the proposed YOLO-vegetable model and the baseline model during the training process. The left subfigure (a) shows the mAP@0.5 curves of YOLO-vegetable and the baseline model. From the figure, it is evident that the proposed model achieves higher mAP during training and converges faster, ultimately reaching 95.6% mAP, significantly outperforming the baseline model&#x2019;s 89.2%. The right subfigure (b) displays the Loss curves of YOLO-vegetable and the baseline model. The baseline model exhibits higher Loss values, slower convergence speed, and ultimately higher final Loss values compared to the proposed model. This indicates that the proposed YOLO-vegetable model not only surpasses the baseline model in accuracy but also demonstrates better convergence behavior and lower loss during training.</p>
<fig id="f14" position="float">
<label>Figure&#xa0;14</label>
<caption>
<p>Comparison results between the proposed YOLO-vegetable model and the baseline model during training process. <bold>(a)</bold> mAP@0.5 curves: YOLO-vegetable (red line) achieves faster convergence and higher final performance (95.6%) compared to baseline model (blue line, 89.2%), demonstrating superior learning capability. <bold>(b)</bold> Loss curves: YOLO-vegetable (red line) exhibits lower loss values and more stable convergence behavior compared to baseline model (blue line), indicating more effective optimization and better model training dynamics.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g014.tif">
<alt-text content-type="machine-generated">Two line graphs comparing the performance of YOLO-vegetable and a baseline model over epochs. Graph (a) shows mean average precision (mAP) at 0.5, where both models improve, with YOLO-vegetable peaking around 90%. Graph (b) displays loss, where YOLO-vegetable decreases more rapidly than the baseline, reaching lower loss values around epoch 20.</alt-text>
</graphic>
</fig>
<p>To comprehensively evaluate the performance of YOLO-vegetable, we conducted extensive comparisons with mainstream object detection models. The experimental results are summarized in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>. On the same vegetable disease dataset, the proposed model exhibits superior comprehensive performance. In terms of detection accuracy, YOLO-vegetable achieves 95.6% mAP@0.5, significantly exceeding the baseline model YOLOv10n (89.2%) and outperforming other mainstream detection algorithms such as Faster-RCNN (89.6%), SSD (94.2%), and YOLOv5s (93.9%). Notably, the proposed model achieves comparable performance to YOLOv10s (95.5%) while demonstrating superior resource efficiency.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Performance comparison with state-of-the-art models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Model</th>
<th valign="top" align="center">mAP@0.5 (%)</th>
<th valign="top" align="center">Parameters (M)</th>
<th valign="top" align="center">FLOPs (G)</th>
<th valign="top" align="center">Inference Time (ms/frame)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">Faster-RCNN</td>
<td valign="middle" align="center">89.6</td>
<td valign="middle" align="center">63.2</td>
<td valign="middle" align="center">370</td>
<td valign="middle" align="center">114.8</td>
</tr>
<tr>
<td valign="bottom" align="center">SSD</td>
<td valign="middle" align="center">94.2</td>
<td valign="middle" align="center">12.3</td>
<td valign="middle" align="center">63.2</td>
<td valign="middle" align="center">22.2</td>
</tr>
<tr>
<td valign="bottom" align="center">YOLOv3</td>
<td valign="middle" align="center">77.8</td>
<td valign="middle" align="center">61.8</td>
<td valign="middle" align="center">43.2</td>
<td valign="middle" align="center">18.9</td>
</tr>
<tr>
<td valign="bottom" align="center">YOLOv5s</td>
<td valign="middle" align="center">93.9</td>
<td valign="middle" align="center">9.1</td>
<td valign="middle" align="center">23.8</td>
<td valign="middle" align="center">17.2</td>
</tr>
<tr>
<td valign="bottom" align="center">YOLOv8s</td>
<td valign="middle" align="center">92.5</td>
<td valign="middle" align="center">11.2</td>
<td valign="middle" align="center">28.5</td>
<td valign="middle" align="center">19.1</td>
</tr>
<tr>
<td valign="bottom" align="center">YOLOv10n</td>
<td valign="middle" align="center">89.2</td>
<td valign="middle" align="center">2.2</td>
<td valign="middle" align="center">6.5</td>
<td valign="middle" align="center">15.6</td>
</tr>
<tr>
<td valign="bottom" align="center">YOLOv10s</td>
<td valign="middle" align="center">94.5</td>
<td valign="middle" align="center">7.2</td>
<td valign="middle" align="center">21.4</td>
<td valign="middle" align="center">24.8</td>
</tr>
<tr>
<td valign="bottom" align="center">YOLOv11n</td>
<td valign="middle" align="center">91.7</td>
<td valign="middle" align="center">2.5</td>
<td valign="middle" align="center">6.3</td>
<td valign="middle" align="center">15.6</td>
</tr>
<tr>
<td valign="bottom" align="center">YOLOv11s</td>
<td valign="middle" align="center">94.0</td>
<td valign="middle" align="center">9.4</td>
<td valign="middle" align="center">21.3</td>
<td valign="middle" align="center">21.8</td>
</tr>
<tr>
<td valign="bottom" align="center">YOLO-vegetable (Ours)</td>
<td valign="middle" align="center">95.6</td>
<td valign="middle" align="center">3.8</td>
<td valign="middle" align="center">14.7</td>
<td valign="middle" align="center">18.6</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From the perspective of model complexity, YOLO-vegetable exhibits significant advantages. Compared to Faster-RCNN&#x2019;s 63.2M parameters, the proposed model requires only 3.8M parameters, reducing storage demands by approximately 94%. In terms of computational efficiency, YOLO-vegetable achieves 14.7 GFLOPs, substantially lower than Faster-RCNN (370.0 GFLOPs) and SSD (63.2 GFLOPs), and also outperforms YOLOv5s (23.8 GFLOPs) and YOLOv8s (28.5 GFLOPs). This marked reduction in computational cost makes the model more suitable for deployment in resource-constrained practical applications.</p>
<p>Regarding real-time performance, YOLO-vegetable achieves an average inference time of 18.6ms per frame, significantly outperforming two-stage detectors such as Faster-RCNN (114.8ms/frame) and single-stage detectors like SSD (22.2ms/frame). Although there is a slight increase compared to the baseline model YOLOv10n (15.6ms/frame), this latency increment is acceptable given the substantial improvement in detection accuracy (from 89.2% to 95.6%). Particularly, compared to YOLOv10s (24.8ms/frame) and YOLOv11s (21.8ms/frame), the proposed model achieves lower inference latency while maintaining comparable detection accuracy, which is crucial for real-time disease monitoring in greenhouse environments.</p>
<p>Through comparisons with various YOLO series variants, it is evident that YOLO-vegetable achieves an optimal balance between performance and lightweight design. Compared to lightweight models such as YOLOv10n (89.2%) and YOLOv11n (91.7%), the proposed model achieves significant accuracy improvements with only moderate increases in parameter count. When compared to YOLOv10s (94.5%) and YOLOv11s (94.0%), it maintains comparable accuracy while substantially reducing model complexity and computational overhead. This balanced performance fully validates the effectiveness of the proposed improvement strategies and provides an efficient and practical solution for vegetable disease detection in real-world applications.</p>
<p>To evaluate the robustness of the proposed model under noisy conditions, we conducted experiments with Gaussian and salt-and-pepper noise. The results demonstrate that YOLO-vegetable maintains high detection accuracy, with mAP@0.5 above 90% in both noise scenarios, highlighting its robustness in real-world applications.</p>
<p>
<xref ref-type="fig" rid="f15">
<bold>Figure&#xa0;15</bold>
</xref> presents representative detection results with bounding boxes across various greenhouse scenarios, including different lighting conditions, planting densities, and disease severities. The visualizations demonstrate YOLO-vegetable&#x2019;s superior detection performance particularly in challenging cases such as partially occluded leaves, early-stage disease symptoms, and complex backgrounds with shadows. Compared to baseline models, our approach shows notably fewer false positives on healthy plant parts with similar color patterns to diseased regions, indicating enhanced feature discrimination capabilities.</p>
<fig id="f15" position="float">
<label>Figure&#xa0;15</label>
<caption>
<p>Detection results with bounding boxes.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1611865-g015.tif">
<alt-text content-type="machine-generated">Six images of plant leaves and stems with visible disease symptoms. Each segment is marked with red boxes containing labels such as A2 94.9, C3 94.7, C4 95.2, A5 95.7, C5 95.8, and B2 95.1, indicating classifications or severity scores.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4_7">
<label>4.7</label>
<title>Generalization experiments</title>
<p>To validate the generalization capability of the proposed YOLO-vegetable model, a public dataset downloaded from the Baidu PaddlePaddle platform in China was selected for generalization testing. This dataset contains 534 images and corresponding annotation files, exhibiting strong scene diversity and challenges. The download link is: <ext-link ext-link-type="uri" xlink:href="https://aistudio.baidu.com/datasetdetail/292158">https://aistudio.baidu.com/datasetdetail/292158</ext-link>. Training was conducted under consistent hardware conditions. The experimental results are shown in <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Generalization experiment results.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Precision (%)</th>
<th valign="middle" align="center">Recall (%)</th>
<th valign="middle" align="center">mAP@0.5 (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">Baseline</td>
<td valign="middle" align="center">86.5</td>
<td valign="middle" align="center">84.2</td>
<td valign="middle" align="center">85.4</td>
</tr>
<tr>
<td valign="middle" align="center">YOLO-vegetable (Ours)</td>
<td valign="middle" align="center">94.8</td>
<td valign="middle" align="center">93.6</td>
<td valign="middle" align="center">94.2</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The experimental results demonstrate that YOLO-vegetable exhibits excellent performance advantages on cross-scene datasets. Compared to the baseline model, precision increased by 8.3% (from 86.5% to 94.8%), indicating that the improved model maintains high detection accuracy in unknown scenarios. Recall increased by 9.4% (from 84.2% to 93.6%), proving that the model has a stronger ability to detect diseases. The mean average precision (mAP@0.5) increased by 8.8% (from 85.4% to 94.2%), demonstrating a significant improvement in the model&#x2019;s overall performance. This comprehensive performance enhancement fully validates the effectiveness of the proposed improvement strategies.</p>
<p>In-depth analysis reveals that the superior generalization performance of YOLO-vegetable is primarily attributed to its enhanced feature representation capability. Through the adaptive detail enhancement mechanism of the ADEConv module, the model can better extract and retain fine-grained features of diseases, enabling accurate recognition across different scenarios. The multi-granularity feature fusion mechanism of the MFLayer allows the model to adaptively handle disease targets of different scales, effectively addressing the issue of target scale variation in cross-scene data. Additionally, the dynamic feature fusion strategy of the IDFNet significantly enhances the model&#x2019;s adaptability to complex backgrounds, ensuring stable detection performance under varying lighting, angles, and occlusion conditions.</p>
<p>Qualitative analysis shows that YOLO-vegetable exhibits clear advantages in handling complex scenarios (e.g., lighting variations, partial occlusion, complex backgrounds), with both false detection and missed detection rates lower than those of the baseline model. This fully demonstrates that the proposed improvement strategies not only enhance the model&#x2019;s detection accuracy but also improve its generalization capability and environmental adaptability. The experimental results indicate that the YOLO-vegetable model has excellent generalization performance, maintaining stable detection performance when faced with new, unseen data, laying a technical foundation for the large-scale application of the model in practical agricultural production.</p>
</sec>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions and future work</title>
<sec id="s5_1" sec-type="conclusions">
<label>5.1</label>
<title>Conclusions</title>
<p>This study successfully addresses critical challenges in greenhouse vegetable disease detection by developing YOLO-vegetable, an enhanced deep learning architecture that significantly improves detection accuracy while maintaining computational efficiency. Our approach represents a substantial advancement in applying AI technology to support agricultural new quality productive forces.</p>
<p>The key contributions of this work include: (1) innovative architectural designs that preserve fine-grained features while enabling multi-scale detection; (2) comprehensive experimental validation demonstrating superior performance across diverse greenhouse conditions; and (3) practical deployment considerations with optimized parameter efficiency. The proposed method achieves state-of-the-art performance on our comprehensive dataset while maintaining real-time capabilities essential for practical applications.</p>
<p>Experimental results validate the effectiveness of our approach, with significant improvements in detection accuracy and computational efficiency compared to existing methods. The model&#x2019;s robust performance across different disease types, growth stages, and environmental conditions demonstrates its potential for widespread adoption in intelligent greenhouse systems. This work provides a foundation for advancing precision agriculture through AI-driven disease monitoring and contributes to the development of sustainable agricultural practices.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Future work</title>
<p>Future research directions include: (1) comprehensive cross-regional validation to establish model generalizability across diverse agricultural settings; (2) development of lightweight architectures for edge computing deployment; (3) integration with IoT systems for automated greenhouse monitoring; and (4) extension to additional crop varieties and disease types. Long-term goals focus on creating comprehensive agricultural AI platforms that support large-scale implementation of intelligent disease management systems in modern farming operations.</p>
</sec>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the article/<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Material</bold>
</xref>.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>JL: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. XW: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. QC: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. PY: Writing &#x2013; review &amp; editing. DG: Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was funded by the Shandong Province Natural Science Foundation (Grant Nos. ZR2023MF048, ZR2023QC116 &amp; ZR2021QC173), the Key R&amp;D Program of Shandong Province, China (Grant No. 2024RZB0206), the Disciplinary Construction Funds of Weifang University of Science and Technology, the Supporting Construction Funds for Shandong Province Data Open Innovation Application Laboratory (KJ&#x2014;C2023001), the School-level Talent Project (Grant No. 2018RC002), the Weifang Soft Science Project (Grant No. 2023RKX184), and the Weifang City Science and Technology Development Plan Project (Grant No. 2023GX051, 2023JH14 &amp; 2024GX033).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. Throughout the preparation of this manuscript, the authors utilized various AI tools to enhance language clarity and readability. Subsequently, the authors meticulously reviewed and edited the content as necessary, assuming full responsibility for the final publication.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpls.2025.1611865/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpls.2025.1611865/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table1.xlsx" id="ST1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"/>
<supplementary-material xlink:href="DataSheet1.csv" id="SM1" mimetype="text/csv"/>
</sec>
<fn-group>
<title>Abbreviations</title>
<fn fn-type="abbr" id="abbrev1">
<p>YOLO, You Only Look Once; ADEConv, Adaptive Detail Enhancement Convolution; MFLayer, Multi-granularity Feature Fusion Detection Layer; IDFNet, Inter-layer Dynamic Fusion Pyramid Network; AAFS, Attention-guided Adaptive Feature Selection; VDD, Vegetable Disease Dataset; mAP, mean Average Precision; IoU, Intersection over Union; CNN, Convolutional Neural Network; PAFPN, Path Aggregation Feature Pyramid Network; BiFPN, Bi-directional Feature Pyramid Network; FPN, Feature Pyramid Network; SPPF, Spatial Pyramid Pooling &#x2013; Fast; PSA, Position-Sensitive Attention.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abdalla</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wheeler</surname> <given-names>T. A.</given-names>
</name>
<name>
<surname>Dever</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Arce</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>W.</given-names>
</name>
</person-group>. (<year>2024</year>). <article-title>Assessing fusarium oxysporum disease severity in cotton using unmanned aerial system images</article-title>. <source>Biosyst. Eng.</source> <volume>237</volume>, <fpage>220</fpage>&#x2013;<lpage>231</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2023.12.014</pub-id>
</citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ali</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>S&#x142;owik</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Hezam</surname> <given-names>I. M.</given-names>
</name>
<name>
<surname>Basset</surname> <given-names>M. A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Sustainable smart system for vegetables plant disease detection: Four vegetable case studies</article-title>. <source>Comput. Electron. Agric.</source> <volume>227</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109672</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alif</surname> <given-names>M. A. R.</given-names>
</name>
<name>
<surname>Hussain</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>YOLOv1 to YOLOv10: A comprehensive review of YOLO variants and their application in the agricultural domain</article-title>. <source>arXiv preprint arXiv:2406.10139</source>.</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bao</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>UAV remote sensing detection of tea leaf blight based on DDMA-YOLO</article-title>. <source>Comput. Electron. Agric.</source> <volume>205</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.107637</pub-id>
</citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barbedo</surname> <given-names>J. G. A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Plant disease identification from individual lesions and spots using deep learning</article-title>. <source>Biosyst. Eng.</source> <volume>180</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2019.02.002</pub-id>
</citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bonora</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Bortolotti</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Bresilla</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Grappadelli</surname> <given-names>L. C.</given-names>
</name>
<name>
<surname>Manfrini</surname> <given-names>L.</given-names>
</name>
</person-group>. (<year>2021</year>). <article-title>A convolutional neural network approach to detecting fruit physiological disorders and maturity in &#x2018;Abb&#xe9; F&#xe9;tel&#x2019; pears</article-title>. <source>Biosyst. Eng.</source> <volume>212</volume>, <fpage>264</fpage>&#x2013;<lpage>272</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2021.10.009</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bouni</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Hssina</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Douzi</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Douzi</surname> <given-names>S.</given-names>
</name>
</person-group>. (<year>2024</year>). <article-title>Synergistic use of handcrafted and deep learning features for tomato leaf disease classification</article-title>. <source>Sci. Rep.</source> <volume>14</volume>, <fpage>26822</fpage>., PMID: <pub-id pub-id-type="pmid">39500934</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Castillo-Girones</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Munera</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Mart&#xed;nez-Sober</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Blasco</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Cubero</surname> <given-names>S.</given-names>
</name>
<name>
<surname>G&#xf3;mez-Sanchis</surname> <given-names>J</given-names>
</name>
</person-group>. (<year>2025</year>). <article-title>Artificial neural networks in agriculture, the core of artificial intelligence: what, when, and why</article-title>. <source>Comput. Electron. Agric.</source> <volume>230</volume>, <fpage>109938</fpage>.</citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Recognition of wheat rusts in a field environment based on improved DenseNet</article-title>. <source>Biosyst. Eng.</source> <volume>238</volume>, <fpage>10</fpage>&#x2013;<lpage>21</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2023.12.016</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>C. F. R.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Panda</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Crossvit: Cross-attention multi-scale vision transformer for image classification</article-title>. <source>In Proc. IEEE/CVF Int. Conf. Comput. Vision</source> <volume>pp</volume>, <fpage>357</fpage>&#x2013;<lpage>366</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00041</pub-id>
</citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chowdhury</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Arko</surname> <given-names>P. S.</given-names>
</name>
<name>
<surname>Ali</surname> <given-names>M. E.</given-names>
</name>
<name>
<surname>Khan</surname> <given-names>M. A. I.</given-names>
</name>
<name>
<surname>Apon</surname> <given-names>S. H.</given-names>
</name>
<name>
<surname>Nowrin</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Identification and recognition of rice diseases and pests using convolutional neural networks</article-title>. <source>Biosyst. Eng.</source> <volume>194</volume>.</citation></ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ding</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Codella</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Davit: Dual attention vision transformers. In European conference on computer vision</source> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>), <fpage>74</fpage>&#x2013;<lpage>92</lpage>.</citation></ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Ghostnet: More features from cheap operations</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. <fpage>1580</fpage>&#x2013;<lpage>1589</lpage>.</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hari</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>M. P.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Adaptive knowledge transfer using federated deep learning for plant disease detection</article-title>. <source>Comput. Electron. Agric.</source> <volume>229</volume>, <fpage>109720</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109720</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Recognition of diseased plants using multi-spectral fusion</article-title>. <source>Biosyst. Eng</source>.</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jian</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>X.</given-names>
</name>
</person-group>. (<year>2025</year>). <article-title>Identification of tomato leaf diseases based on DGP-SNNet</article-title>. <source>Crop Prot.</source> <volume>187</volume>, <elocation-id>106975</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cropro.2024.106975</pub-id>
</citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johri</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Dixit</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Kakkar</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Advanced deep transfer learning techniques for efficient detection of cotton plant diseases</article-title>. <source>Front. Plant Sci.</source> <volume>15</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpls.2024.1441117</pub-id>, PMID: <pub-id pub-id-type="pmid">39759238</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Toward real scenery: A lightweight tomato growth inspection algorithm for leaf disease detection and fruit counting</article-title>. <source>Plant phenomics</source> <volume>6</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/plantphenomics.0174</pub-id>, PMID: <pub-id pub-id-type="pmid">38629080</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karantoumanis</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Balafas</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Louta</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ploskas</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Real-time disease detection on bean leaves from a small image dataset using data augmentation and deep learning methods</article-title>. <source>Soft Computing</source> <volume>28</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00500-024-10348-3</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kumar</surname> <given-names>V. S.</given-names>
</name>
<name>
<surname>Jaganathan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Viswanathan</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Umamaheswari</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vignesh</surname> <given-names>J. J. E. R. C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Rice leaf disease detection based on bidirectional feature attention pyramid network with YOLO v5 model</article-title>. <source>Environ. Res. Commun.</source> <volume>5</volume>, <fpage>065014</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1088/2515-7620/acdece</pub-id>
</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Mixed data augmentation and osprey search strategy for enhancing YOLO in tomato disease, pest, and weed detection</article-title>. <source>Expert Syst. With Appl.</source> <volume>264</volume>.</citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Min</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Mei</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Plant disease recognition: A large-scale benchmark dataset and a visual region and loss reweighting approach</article-title>. <source>IEEE Trans. Image Process.</source> <volume>30</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.83</pub-id>, PMID: <pub-id pub-id-type="pmid">33444137</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>EFDet: An efficient detection method for cucumber disease under natural complex environments</article-title>. <source>Comput. Electron. Agric</source>.</citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mathieu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Reder</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Siah</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ducasse</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Langlands-Perry</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Marcel</surname> <given-names>T. C.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>SeptoSympto: a precise image analysis of Septoria tritici blotch disease symptoms using deep learning methods</article-title>. <source>Plant Methods</source> <volume>20</volume>, <fpage>18</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s13007-024-01136-z</pub-id>, PMID: <pub-id pub-id-type="pmid">38297386</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mhala</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Bilandani</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Enhancing crop productivity with fined-tuned deep convolution neural network for Potato leaf disease detection</article-title>. <source>Expert Syst. With Appl.</source> <volume>267</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2024.126066</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mo</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Lightweight citrus leaf disease detection model based on ARMS and cross-domain dynamic attention</article-title>. <source>J. King Saud Univ. - Comput. Inf. Sci.</source> <volume>36</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jksuci.2024.102133</pub-id>
</citation></ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paul</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Sunil</surname> <given-names>G. C.</given-names>
</name>
<name>
<surname>Horvath</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Deep learning for plant disease detection: A comprehensive review of technologies, challenges, and future directions</article-title>. <source>Comput. Electron. Agric.</source> <volume>229</volume>.</citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qing</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>GPT-aided diagnosis on agricultural image based on a new light YOLOPC</article-title>. <source>Comput. Electron. Agric.</source> <volume>213</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108168</pub-id>
</citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Si</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>Research on tomato disease image recognition method based on DeiT</article-title>. <source>Eur. J. Agron.</source> <volume>162</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eja.2024.127400</pub-id>
</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Pang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Le</surname> <given-names>Q. V.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Efficientdet: Scalable and efficient object detection</article-title>. <source>Proc. IEEE/CVF Conf. Comput. Vision Pattern recognition</source>, <fpage>10781</fpage>&#x2013;<lpage>10790</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR42600.2020</pub-id>
</citation></ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tian</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>VMF-SSD: A novel V-space based multi-scale feature fusion SSD for apple leaf disease detection</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinf</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TCBB.2022.3229114</pub-id>, PMID: <pub-id pub-id-type="pmid">37015544</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Toda</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Okura</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>How convolutional neural networks diagnose plant disease</article-title>. <source>Plant Phenomics</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.34133/2019/9237136</pub-id>, PMID: <pub-id pub-id-type="pmid">33313540</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Upadhyay</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chandel</surname> <given-names>N. S.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>K. P.</given-names>
</name>
<name>
<surname>Chakraborty</surname> <given-names>S. K.</given-names>
</name>
<name>
<surname>Nandede</surname> <given-names>B. M.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>Deep learning and computer vision in plant disease detection: a comprehensive review of techniques, models, and trends in precision agriculture</article-title>. <source>Artif. Intell. Rev.</source> <volume>58</volume>, <fpage>1</fpage>&#x2013;<lpage>64</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10462-024-11100-x</pub-id>
</citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>V&#xe1;sconez</surname> <given-names>J. P.</given-names>
</name>
<name>
<surname>V&#xe1;sconez</surname> <given-names>I. N.</given-names>
</name>
<name>
<surname>Moya</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Calder&#xf3;n-D&#xed;az</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Valenzuela</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Besoain</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Deep learning-based classification of visual symptoms of bacterial wilt disease caused by Ralstonia solanacearum in tomato plants</article-title>. <source>Comput. Electron. Agric.</source> <volume>227</volume>, <elocation-id>109617</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2024.109617</pub-id>
</citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Yolov10: Real-time end-to-end object detection</article-title>. <source>arXiv preprint arXiv:2405.14458</source>.</citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>He</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>WCG-VMamba: A multi-modal classification model for corn disease</article-title>. <source>Comput. Electron. Agric.</source> <volume>230</volume>.</citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>W&#xf3;jcik Gront</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Zieniuk</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Pawe&#x142;kowicz</surname> <given-names>M.</given-names>
</name>
</person-group>. (<year>2024</year>). <article-title>Harnessing AI-powered genomic research for sustainable crop improvement</article-title>. <source>Agriculture</source> <volume>14</volume>, <elocation-id>2299</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture14122299</pub-id>
</citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Kong</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Xing</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Cong</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Real-time object detection method of melon leaf diseases under complex background in greenhouse</article-title>. <source>J. Real-Time Image Process.</source> <volume>19</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11554-022-01239-7</pub-id>
</citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Deep transfer learning for cross-species plant disease diagnosis adapting mixed subdomains</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinf</source>., PMID: <pub-id pub-id-type="pmid">34914593</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Plant disease detection with vision-language fusion framework</article-title>. <source>Comput. Electron. Agriculture</source>.</citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ye</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Shao</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Detection model of tea disease severity under low light intensity based on YOLOv8 and enlightengan</article-title>. <source>Plants</source> <volume>13</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/plants13101377</pub-id>, PMID: <pub-id pub-id-type="pmid">38794447</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>H. S.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W. F.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X. G.</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>C. Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Enhancing wheat Fusarium head blight detection using rotation Yolo wheat detection network</article-title>. <source>Comput. Electron. Agric.</source> <volume>211</volume>, <fpage>107968</fpage>.</citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Detection of wheat scab fungus spores utilizing the Yolov5-ECA-ASFF network structure</article-title>. <source>Comput. Electron. Agric.</source> <volume>210</volume>.</citation></ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>Plant disease detection using generated leaves based on doubleGAN</article-title>. <source>IEEE/ACM Trans. Comput. Biol. Bioinf</source>., PMID: <pub-id pub-id-type="pmid">33534712</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Taha</surname> <given-names>M. F.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Phenotypic analysis of diseased plant leaves using supervised and weakly supervised deep learning</article-title>. <source>Plant Phenomics</source> <volume>5</volume>.</citation></ref>
</ref-list>
</back>
</article>