<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1632698</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A sorghum seed variety identification method based on image&#x2013;hyperspectral fusion and an improved deep residual convolutional network</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Yang</surname>
<given-names>Xu</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2666940/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Yihan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3071105/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Song</surname>
<given-names>Shaozhong</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2918921/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Zhimin</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3156430/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Electronic Information Engineering, Changchun University of Science and Technology</institution>, <addr-line>Changchun</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Data Science and Artificial Intelligence, Jilin Engineering Normal University</institution>, <addr-line>Changchun</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Jilin Academy of Agricultural Sciences Peanut Institute</institution>, <addr-line>Gongzhuling, Jilin</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Alejandro Isabel Luna-Maldonado, Autonomous University of Nuevo Le&#xf3;n, Mexico</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Jing Yao, Chinese Academy of Sciences (CAS), China</p>
<p>Ren Pengju, Shanghai Maritime University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Shaozhong Song, <email xlink:href="mailto:songsz@jlenu.edu.cn">songsz@jlenu.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>22</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1632698</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Yang, Chen, Song and Zhang.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Yang, Chen, Song and Zhang</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Sorghum is an important food and feed crop. Identifying sorghum seed varieties is crucial for ensuring seed quality, improving planting efficiency, and promoting sustainable agricultural development.</p>
</sec>
<sec>
<title>Methods</title>
<p>This study proposes a high-precision classification method based on the fusion of RGB images and hyperspectral data, using an improved deep residual convolutional neural network. A spectrogram fusion dataset containing 12,800 seeds from eight sorghum varieties was constructed. The network was enhanced by integrating depthwise separable convolution (DSC) and the Convolutional Block Attention Module (CBAM) into the ResNet50 framework.</p>
</sec>
<sec>
<title>Results</title>
<p>The CBAM-ResNet50-DSC model demonstrated outstanding performance, achieving a classification accuracy of 94.84%, specificity of 99.20%, recall of 94.39%, precision of 94.52%, and an F1-score of 0.9438 on the fusion dataset.</p>
</sec>
<sec>
<title>Discussion</title>
<p>These results confirm that the proposed model can accurately and non-destructively classify sorghum seed varieties. The method offers a dependable and efficient approach for seed screening and has practical value in agricultural applications.</p>
</sec>
</abstract>
<kwd-group>
<kwd>artificial intelligence</kwd>
<kwd>sorghum seed</kwd>
<kwd>variety identification</kwd>
<kwd>multi-modal fusion</kwd>
<kwd>ResNet models</kwd>
</kwd-group>
<counts>
<fig-count count="17"/>
<table-count count="6"/>
<equation-count count="5"/>
<ref-count count="29"/>
<page-count count="18"/>
<word-count count="8855"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Sorghum is a drought and heat-tolerant cereal crop widely used for food, feed, and brewing. Its gluten-free nature and associated health benefits have also drawn increasing attention in developing functional and health-oriented food products (<xref ref-type="bibr" rid="B13">Khoddami et&#xa0;al., 2023</xref>). Seed purity refers to the consistency of sorghum seeds in maintaining their characteristic traits, which directly impacts the crop&#x2019;s yield and quality. During harvesting and storage, impurities may be unintentionally mixed into seed lots, leading to economic losses in agricultural production and processing. Moreover, some individuals or companies may intentionally substitute inferior sorghum seeds for high-quality varieties in the seed market to gain additional profit (<xref ref-type="bibr" rid="B29">Zhang et&#xa0;al., 2017</xref>). Therefore, developing a rapid and nondestructive detection technique to screen and grade sorghum seeds before they enter the market is essential, ensuring effective agricultural production implementation, quality control, and market supervision.</p>
<p>Traditional methods for seed identification include visual inspection (<xref ref-type="bibr" rid="B18">Shahin et&#xa0;al., 2006</xref>), flotation (<xref ref-type="bibr" rid="B21">Torchio et&#xa0;al., 2014</xref>), microscopic analysis (<xref ref-type="bibr" rid="B3">Asmussen et&#xa0;al., 2015</xref>), chemical testing (<xref ref-type="bibr" rid="B5">Chen et&#xa0;al., 2016</xref>), and germination experiments (<xref ref-type="bibr" rid="B2">Andersson et&#xa0;al., 2007</xref>). Although these approaches are simple and easy to implement, they are time-consuming and highly subjective, making them insufficient to meet the demands of modern agriculture. Therefore, there is an urgent need for a rapid, accurate, and nondestructive method for identifying and classifying sorghum seeds.</p>
<p>In recent years, image processing and deep learning approaches have received a lot of interest in the subject of seed classification. For example, Franco C showed that by combining data augmentation with deep convolutional neural networks, seed vigor could be predicted with up to 90% accuracy using simple RGB information, such as shape, color, and size (<xref ref-type="bibr" rid="B7">Franco et&#xa0;al., 2020</xref>). Similarly, Masuda K used deep learning and interpretable AI approaches to perform noninvasive diagnosis of seedless/seeded internal features in persimmon fruits, with a VGG16 model classification accuracy of up to 89% from simple RGB photos (<xref ref-type="bibr" rid="B16">Masuda et&#xa0;al., 2021</xref>). In another study, Sunil, G used the VGG16 deep learning classifier to classify RGB images of four weeds (horseshoe grass, kochia, ragweed, and waterhemp) and six crops (black beans, canola, corn, flax, soybeans, and sugar beets). The results showed that the average Fl score of the VGG16 model classifier ranged between 93% and 97.5 (<xref ref-type="bibr" rid="B20">Sunil et&#xa0;al., 2022</xref>).</p>
<p>In summary, RGB data collected by industrial cameras, paired with deep learning models, performed well in seed classification. However, relying solely on RGB photos does not completely utilize the spectral information inside the seeds, resulting in certain limitations in categorization accuracy. As a result, hyperspectral imaging technology is a cutting-edge technology that has advanced rapidly in recent years, and it has been integrated with artificial intelligence algorithms to create a new nondestructive detection technique (<xref ref-type="bibr" rid="B12">Kamruzzaman et&#xa0;al., 2016</xref>). For example, Soares and SFC employed near-infrared hyperspectral imaging (NIR-HSI) to quickly and non-destructively classify cotton seed varieties. The NIR-HSI, conventional NIR, and conventional VIS-NIR datasets were correctly classified at 98.0%, 89.7%, and 91.7%, respectively, using partial least squares discriminant analysis (<xref ref-type="bibr" rid="B19">Soares et&#xa0;al., 2016</xref>). Similarly, An, JL introduced a unique feature extraction method called Low-Rank Tensor Approximation (LRTA) based on hyperspectral images, which improved accuracy by 4% over the old method (<xref ref-type="bibr" rid="B1">An et&#xa0;al., 2023</xref>). In another study, Malik showed that integrating hyperspectral imaging (HSI) and convolutional neural networks (CNNs) could quickly and non-destructively estimate the tofu quality of soybean seeds with 96-99% accuracy (<xref ref-type="bibr" rid="B15">Malik et&#xa0;al., 2024</xref>). Recently, Yao et&#xa0;al. proposed Spectral Mamba, an efficient state-space model for hyperspectral image classification (<xref ref-type="bibr" rid="B28">Yao et&#xa0;al., 2024</xref>), while Pang et&#xa0;al. introduced SPECIAL, a CLIP-based zero-shot classification framework that eliminates the need for manual annotations (<xref ref-type="bibr" rid="B17">Pang et&#xa0;al., 2025</xref>), providing new directions for efficient and generalizable HSI analysis.</p>
<p>In recent years, to enhance feature extraction capability and computational efficiency in agricultural image analysis, attention mechanisms (such as SE, ECA, and CBAM) and depthwise separable convolutions (DSC) have been widely introduced into various detection and classification tasks. Jiang et&#xa0;al. proposed a deep learning-based method for dense Muscovy duck detection. By integrating CBAM modules into the YOLOv7 framework, they developed the CBAM-YOLOv7 model. Experimental results demonstrated that this method outperformed SE-YOLOv7 and ECA-YOLOv7 in terms of accuracy, recall, and mAP, confirming the effectiveness of attention mechanisms in dense livestock detection tasks (<xref ref-type="bibr" rid="B11">Jiang et&#xa0;al., 2022</xref>). Guo et&#xa0;al. proposed an improved SSD-based method for cotton leaf disease detection to address the problems of large model size and low detection accuracy. By introducing the lightweight MobileNetV2 as the backbone and integrating SE, ECA, and CBAM attention mechanisms, the model significantly reduced parameters and computation while enhancing detection speed and accuracy. Among the variants, the SSD_MobileNetV2+ECA model achieved the highest precision, recall, F1-score, mAP, and FPS, demonstrating that attention mechanisms can effectively enhance feature representation and improve detection performance under complex conditions (<xref ref-type="bibr" rid="B8">Guo et&#xa0;al., 2024</xref>). Tyagi et&#xa0;al. proposed a hyperspectral imaging approach combined with improved depthwise separable convolution to assess fruit maturity. Applied to kiwifruit and avocado, the model achieved higher accuracy in predicting maturity, firmness, and sugar content, outperforming state-of-the-art methods (<xref ref-type="bibr" rid="B23">Tyagi et&#xa0;al., 2024</xref>). In summary, although existing studies have made progress in seed classification using RGB images and hyperspectral techniques, and methods such as attention mechanisms and depthwise separable convolutions have shown potential in enhancing model performance and efficiency, each approach has its limitations. RGB-based methods lack internal spectral information, while hyperspectral approaches often face high data complexity. Moreover, multi-modal studies specifically targeting sorghum seeds remain limited, highlighting the need for further research to improve classification accuracy and practical applicability. As a result, our project team has previously investigated the merging of geometric and textural features taken from photos with hyperspectral data to create multi-modal feature vectors and categorize them using machine learning algorithms. The results demonstrate that multi-modal data fusion can significantly increase classification performance (<xref ref-type="bibr" rid="B4">Bi et al., 2024</xref>). However, in real applications, the preprocessing procedure of integrating multi-modal data into one-dimensional vectors is complex, increasing the preparation workload. As a result, this paper presents a novel multi-modal fusion technique and employs an upgraded ResNet network model for classification.</p>
<p>The main contributions and novelties of this work are summarized as follows:</p>
<list list-type="order">
<list-item>
<p>We propose a novel data-level multi-modal fusion strategy that transforms one-dimensional hyperspectral data into two-dimensional reflectance curve images and concatenates them with RGB images to form a unified spectrogram-like input. This early-stage data fusion preserves both spatial and spectral characteristics in a structurally consistent format (224*224*3), enabling the network to extract complementary information more effectively and facilitating end-to-end learning.</p>
</list-item>
<list-item>
<p>We design an enhanced ResNet50-based classification model by integrating the Convolutional Block Attention Module (CBAM), which strengthens the model&#x2019;s ability to focus on critical spatial and channel features, leading to improved feature discrimination and classification performance.</p>
</list-item>
<list-item>
<p>To further reduce model complexity and improve computational efficiency, we incorporate depthwise separable convolution (DSC) into the network. This not only lowers the number of parameters and FLOPs but also maintains high accuracy, making the model more suitable for large-scale or resource-constrained applications.</p>
</list-item>
</list>
<p>Overall, this study introduces a lightweight and effective framework for high-resolution sorghum seed classification, demonstrating the value of data-level fusion in enhancing feature representation and model performance.</p>
<p>The following of this article was organized as the section &#x201c;Materials and Methods&#x201d; described the details of the datasets and the overview of the methods, the experimental results were described and discussed in the section &#x201c;Results and Discussions,&#x201d; and the section &#x201c;Conclusions&#x201d; was the concluding remarks.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Image acquisition and preprocessing</title>
<sec id="s2_1_1">
<label>2.1.1</label>
<title>Data source and acquisition</title>
<p>The Jilin Academy of Agricultural Sciences, Jilin Province, supplied eight distinct sorghum seed varieties, including JZ127, JZ136, JZ141, JZ159, JZ160, JZ177, JZ186, and JZ187, which were employed in this experiment. There were 12,800 seeds in total, 1600 seeds in each type. Training, validation, and testing were the three groups into which the dataset was split in a 7:2:1 ratio.</p>
</sec>
<sec id="s2_1_2">
<label>2.1.2</label>
<title>Data acquisition and preprocessing</title>
<sec id="s2_1_2_1">
<label>2.1.2.1</label>
<title>RGB image data</title>
<p>A Nikon camera (Nikon D7100) was used to take RGB pictures of these seeds, and <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1a</bold>
</xref> displays the RGB capture scheme. The seeds were carefully chosen and validated by professionals before images were taken to guarantee that the samples were entire, consistently shaped, and free of dust and contaminants. Every seed that was chosen acted normally looked tidy, and showed no signs of damage. For imaging, 1600 samples of each type were randomly picked; the sample size was selected to accommodate the substantial data needed for the deep learning model (<xref ref-type="bibr" rid="B25">Wen, 2020</xref>). <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1b</bold>
</xref> displays the RGB pictures of the gathered sorghum seeds.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>
<bold>(a)</bold> Schematic diagram of the RGB imaging setup for sorghum seeds <bold>(b)</bold> RGB images of eight sorghum seed varieties.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g001.tif">
<alt-text content-type="machine-generated">Illustration shows an industrial setup featuring a computer connected to two cameras and illumination lamps aimed at a platform labeled 'Sorghum Seeds'. The right side displays eight sorghum seeds arranged in two rows, highlighting variations in their appearance.</alt-text>
</graphic>
</fig>
<p>The main goal of sorghum variety identification is to ensure sorghum seeds are pure, particularly to confirm the legitimacy of individual seeds. An image with several seeds must be segmented to employ a single seed recognition technique to differentiate between various sorghum seed types. The original image is first transformed to greyscale to highlight brightness-related elements and exclude color information. A binarized image is then produced by applying automatic global thresholding and morphological filtering procedures, simplifying the image and extracting the target contours. Lastly, morphological filtering was used to score and mask the sorghum seed area in the binarized image. After that, it was divided into separate 224*224 sorghum seeds, yielding 12,800 raw photos. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> depicts the picture-cutting procedure.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Sorghum seed image cutting preprocessing.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g002.tif">
<alt-text content-type="machine-generated">A sequential image showing stages of seed image processing. Top left: original image with seeds on a dark background. Top right: grayscale version of the original. Bottom right: binary mask highlighting seeds. Bottom left: individual images of separated seeds. Arrows indicate processing flow from original to each step.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_1_2_2">
<label>2.1.2.2</label>
<title>Hyperspectral data</title>
<p>The spectral data of sorghum seeds was obtained in this experiment using a FieldSpec4 ground spectrometer; <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3a</bold>
</xref> displays the schematic diagram of the bright light data acquisition. The instrument operates in the visible (VNIR), near-infrared (NIR), and short-wave infrared (SWIR) bands, which span the wavelength range of 350&#x2013;2500 nm. It is ideal for fine spectrum analysis due to its broad wavelength range, high signal-to-noise ratio (SNR up to 9000:1), and excellent spectral resolution (VNIR 1&#x2013;3 nm, NIR 3&#x2013;5 nm, SWIR 5&#x2013;10 nm). In this experiment, the spectral characteristics of sorghum seeds may be efficiently characterized for seed classification utilizing hyperspectral data obtained with the FieldSpec4 ground spectrometer. Because of its structure and effectiveness, a one-dimensional form is typically employed to store spectral data. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3b</bold>
</xref> displays the one-dimensional spectrum data.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>
<bold>(a)</bold> Schematic diagram of the hyperspectral data acquisition setup for sorghum seeds <bold>(b)</bold> Schematic representation of the raw spectral data storage format for sorghum seeds.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g003.tif">
<alt-text content-type="machine-generated">Diagram labeled (a) shows a setup with an optical fiber, pistol, field of view, halogen tungsten lamps, and sorghum seeds connected to computer equipment, including a Fieldspec4. Diagram labeled (b) presents a dataset of spectral measurements ranging from 350 to 2500 with varying numerical values.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_1_2_3">
<label>2.1.2.3</label>
<title>2D fusion of image data with hyperspectral data</title>
<p>In earlier studies, this project team successfully built multi-modal feature vectors, applied machine learning techniques for classification, and merged geometric and textural aspects of images with hyperspectral data. The findings demonstrate that the classification performance is much enhanced by multi-modal data fusion. The multi-modal data must be combined into one-dimensional vectors using this fusion approach, and the preprocessing step is challenging, adding to the data processing workload. This paper suggests a novel approach to data fusion: upscaling one-dimensional spectral data to two-dimensional curves and fusing them with two-dimensional picture data. This approach increases the fusion process&#x2019;s efficiency while streamlining the data preprocessing step. In particular, wavelengths and their related reflectance values are employed to record the spectral data, with the wavelength serving as the horizontal coordinate and the reflectance as the vertical coordinate. The one-dimensional spectral data are plotted intuitively as spectral reflectance curves using data visualization tools (such as Python&#x2019;s matplotlib library). Using Python visualization tools such as matplotlib, the processed spectral data are plotted into intuitive reflectance curves, clearly reflecting the sample&#x2019;s spectral response across different wavelengths and providing a solid foundation for further data analysis and fusion. <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4a</bold>
</xref> illustrates a typical spectral curve of collected sorghum seeds. To enhance curve smoothness and readability, Savitzky-Golay (SG) filtering is applied prior to plotting, effectively suppressing high-frequency noise while preserving key structural features, as shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4b</bold>
</xref>. Given that this study fuses spectral and RGB image data in image format for classification, more complex preprocessing techniques such as SNV or MSC normalization-which may distort curve shapes or introduce redundancy were deliberately avoided. Instead, SG filtering is selected as the sole preprocessing method due to its simplicity, low computational cost, and excellent shape-preserving capability, ensuring the physical and visual integrity of the spectral data for image fusion and model input.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>
<bold>(a)</bold> Raw hyperspectral curves <bold>(b)</bold> Preprocessed hyperspectral curves <bold>(c)</bold> Fusion of RGB data and hyperspectral data.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g004.tif">
<alt-text content-type="machine-generated">Two graphs labeled (a) and (b) show original and SG smoothed spectral curves, respectively, with multiple colored lines indicating reflectance over wavelengths. Below, part (c) shows a visual representation of a fruit combined with a hyperspectral curve, resulting in a fusion image and corresponding graph.</alt-text>
</graphic>
</fig>
<p>Subsequently, a new two-dimensional spectral fusion dataset was constructed by horizontally concatenating the RGB images with the preprocessed spectral curve images. As shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4c</bold>
</xref>, the original RGB image with size 224*224*3 and the grayscale spectral curve image with size 224*224*1 were first aligned in format. To address the channel mismatch, the grayscale image was replicated across the R, G, and B channels to form a pseudo-color image with size 224*224*3, maintaining the visual appearance while ensuring structural compatibility. The two images were then stitched side by side to form a unified fusion image with a final resolution of 224*448*3 This fusion strategy preserves the intuitive visual information of the RGB image while incorporating the key spectral features from the hyperspectral data. It ensures structural consistency in the fused input and enhances the expressive power of the data, providing a richer and more integrated multi-modal representation for deep learning-based classification.</p>
</sec>
</sec>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Building the model</title>
<p>Gradient degradation and gradient vanishing issues frequently arise during the training phase of convolutional neural networks (CNNs) as their depth grows, impacting the model&#x2019;s convergence rate and ultimate accuracy (<xref ref-type="bibr" rid="B26">Wenchao and Zhi, 2022</xref>). Although the gradient vanishing and the ResNet family of networks somewhat mitigate explosion issues in deep neural networks, performance deterioration may still occur during deep model training (<xref ref-type="bibr" rid="B22">Traore et&#xa0;al., 2018</xref>). This paper introduces the attention mechanism into the network structure to improve sorghum seeds&#x2019; recognition performance and classification accuracy. It replaces some standard convolutions with depth-separable convolutions to enhance the model&#x2019;s ability to extract key information. In order to create a fast and nondestructive variety classification method based on sorghum seed image data and hyperspectral data, this experiment will use deep learning algorithms for eight different types of sorghum seeds (ResNet18, ResNet34, ResNet50, ResNet101, SENet-ResNet50, CBAM-ResNet50, ECA-ResNet50, CBAM-ResNet50-DSC, eight residual network models).</p>
<sec id="s2_2_1">
<label>2.2.1</label>
<title>ResNet model</title>
<p>(<xref ref-type="bibr" rid="B9">He et&#xa0;al., 2016</xref>) proposed the deep neural network structure known as ResNet. While increasing the network layers, this network successfully addresses the gradient vanishing issue and enhances parameter consumption efficiency by implementing the residual connection mechanism. ResNet is a popular model choice for sorghum seed detection tasks because of its strong learning ability for complicated features and rapid inference speed, which allows it to perform better while maintaining good generalization performance. The Residual Block, the fundamental unit structure of the residual network, is seen in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5a</bold>
</xref>. The structure uses a Skip Connection and a primary path to implement feature learning and transfer. By applying two convolutional transforms to the input feature x and utilizing the ReLU activation function following each convolution, the main path determines the residual F(x). The input feature x is then sent straight to the output by the bypass connection, where it is combined with the residuals F(x) that the primary path has learned to create the final output F(x)+x. This design successfully increases the network&#x2019;s performance capability and training efficiency by preserving the information of the input features and resolving the gradient vanishing issue in the deep network. The residual block is the fundamental building element of ResNet, which introduces skip connections and identity mapping to address the gradient vanishing and gradient explosion issues in deep neural networks. <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5b</bold>
</xref> illustrates the overall architecture of the residual network using ResNet50 as an example. This network adopts a fully convolutional structure and does not include any fully connected layers that enforce fixed input dimensions during the feature extraction stage, thereby exhibiting strong adaptability to varying input sizes. As long as the input image maintains valid spatial dimensions after multiple convolution and pooling operations, the network can operate stably with consistent output structure. Accordingly, this study uses a concatenated input image with dimensions of 224*224*3, which can be directly fed into the ResNet50 model for feature extraction and classification without any structural modifications. In the first stage, a 7*7 convolutional layer followed by a 3*3 max pooling layer downsamples the input to reduce its spatial resolution. Then, in the second stage, the residual modules Conv2, Conv3, Conv4, and Conv5 are introduced sequentially to extract higher-level semantic features. Throughout this process, the spatial dimensions of the feature maps gradually decrease from 56 to 7 in height and from 112 to 14 in width, while the number of channels increases from 256 to 2048. In the third stage, global average pooling compresses the feature map to 1*1*2048, and the final classification result is obtained through a fully connected layer.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Structure of the residual block and structure of the Resnet50 network. <bold>(a)</bold> the residual block <bold>(b)</bold> ResNet50 structure diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g005.tif">
<alt-text content-type="machine-generated">Diagram illustrating a neural network architecture. On the left, a residual block with two weight layers and ReLU activation, showing identity mapping. On the right, the main architecture includes five convolutional stages. Stage 1 starts with a convolution layer followed by batch normalization, ReLU, and max pooling. Stages 2 to 5 consist of convolutional blocks with increasing complexity. The final stage includes average pooling, a fully connected layer, and a softmax output, highlighting the transition from input to output dimensions.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_2_2">
<label>2.2.2</label>
<title>Attention mechanisms</title>
<sec id="s2_2_2_1">
<label>2.2.2.1</label>
<title>Squeeze-and-excitation networks</title>
<p>SENet (Squeeze-and-Excitation Networks) is a deep learning model based on the attention mechanism, which dynamically models the channel relationship of feature maps by introducing the Squeeze-and-Excitation (SE) module to enhancing the attention to the essential features and suppressing the irrelevant features (<xref ref-type="bibr" rid="B14">Li et&#xa0;al., 2020</xref>). The input feature maps are constantly adjusted by SENet&#x2019;s Squeeze-and-Excitation module in three stages: Squeeze, Excitation, and Reweighting. The global description of each channel is first obtained in the Squeeze stage by using Global Average Pooling (GAP) to compress the input feature map <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the spatial dimension. In particular, each channel&#x2019;s global feature <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is calculated using <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>.</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>H</mml:mi>
</mml:munderover>
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>H</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>A vector with dimensions of 1&#xd7;1&#xd7;C is the end product.</p>
<p>A two-layer Fully Connected Network (FC) captures the nonlinear interaction between channels and generates dynamic channel weights in the Excitation step. The nonlinear changes are introduced using the ReLU activation function after the first layer of the fully linked network decreases the number of channels to <inline-formula>
<mml:math display="inline" id="im3">
<mml:mi>r</mml:mi>
</mml:math>
</inline-formula> times the initial number (often <inline-formula>
<mml:math display="inline" id="im4">
<mml:mi>r</mml:mi>
</mml:math>
</inline-formula> = 16, or dimensionality reduction). The second layer of the completely linked network then uses the Sigmoid activation function to create the normalized weight s, which has a size of <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and returns the number of channels to its initial size.</p>
<p>The final output feature map is created by multiplying the generated channel weights by the input feature map channel by the channel during the Reweighting stage. In particular, the output feature <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>c</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is calculated as shown in <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi>X</mml:mi>
<mml:mi>c</mml:mi>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>X</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where the weight of channel <inline-formula>
<mml:math display="inline" id="im7">
<mml:mi>c</mml:mi>
</mml:math>
</inline-formula> is denoted by <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow><mml:mo>&#x200b;</mml:mo>
<mml:mi>c</mml:mi></mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The SE module can enhance the expressiveness and classification performance of the model by emphasizing significant features and suppressing unimportant ones through a dynamic weighting technique. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> displays the SE module diagram.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>SE module diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g006.tif">
<alt-text content-type="machine-generated">Diagram of a SeNet module showing a process flow: input is a 3D block labeled &#x201c;h&#x201d;, &#x201c;w&#x201d;, and &#x201c;c sub 2&#x201d;. It undergoes a &#x201c;Squeeze&#x201d; step reducing to a 1D vector, followed by an &#x201c;Excitation&#x201d; step. The output is a scaled 3D block with layered colors, labeled the same way as input. Arrows depict the flow direction.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_2_2_2">
<label>2.2.2.2</label>
<title>Convolutional block attention module</title>
<p>CBAM (Convolutional Block Attention Module) is a lightweight and efficient attention mechanism that can significantly improve the performance of Convolutional Neural Networks (<xref ref-type="bibr" rid="B27">Woo et&#xa0;al., 2018</xref>). Applying channel and spatial attention to the input feature maps highlights significant channels and crucial spatial locations. While the Spatial Attention module creates spatial weights using pooling and convolution operations, the Channel Attention module uses global average pooling and maximum pooling to extract global context information and build channel weights. In <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>, the CBAM structure is displayed.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>CBAM module diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g007.tif">
<alt-text content-type="machine-generated">Flow chart illustrating the Convolutional Block Attention Module (CBAM). It begins with an &#x201c;Input Feature&#x201d; represented as a box, followed by a &#x201c;Channel Attention Module&#x201d; and a multiplication symbol. This process proceeds to a &#x201c;Spatial Attention Module&#x201d; with another multiplication symbol, resulting in a &#x201c;Refined Feature&#x201d; box. Arrows indicate the flow direction.</alt-text>
</graphic>
</fig>
<p>The output <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> of the channel attention module can be calculated using <xref ref-type="disp-formula" rid="eq3">Equation 3</xref>:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>MLP stands for multilayer perceptual machine, &#x3c3; for sigmoid activation function, F for input feature map, and AvgPool and MaxPool for global average pooling and maximum pooling operations, respectively, in Eq.</p>
<p>And the loss <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> of the spatial attention module can be calculated by <xref ref-type="disp-formula" rid="eq4">Equation 4</xref>:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>
<inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> indicates sewing together the average pooling and maximum pooling results along the channel axis, whereas <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mn>7</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> indicates a 7&#xd7;7 convolution operation.</p>
</sec>
<sec id="s2_2_2_3">
<label>2.2.2.3</label>
<title>Efficient channel attention</title>
<p>ECA (Efficient Channel Attention) is an effective channel attention method that dramatically lowers the computational complexity by eliminating the fully connected layer from the conventional SE module and utilizing 1D convolution to capture the local interactions between channels (<xref ref-type="bibr" rid="B24">Wang et&#xa0;al., 2019</xref>). Using an adjustable convolution kernel size that dynamically interacts with the number of channels, ECA can guarantee that the model is lightweight while significantly enhancing network performance. The construction of ECA is depicted in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, where a channel description vector <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>C</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is obtained by first undergoing Global Average Pooling (GAP) on the input feature map <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>H</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>W</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and then extracting the global semantic information of each channel. Instead of using a fully connected layer, ECA dynamically determines the size k of the one-dimensional convolution kernel based on the number of channels <italic>C</italic>, which is calculated using <xref ref-type="disp-formula" rid="eq5">Equation 5</xref>. This allows for the realization of local cross-channel interactions without dimensionality compression and, in the end, generates the channel attention weights. This allows for the efficient modeling of the interrelationships between channels.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>ECA module diagram.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g008.tif">
<alt-text content-type="machine-generated">Diagram of the &#x201c;efficient channel attention module&#x201d; structure. It shows input dimensions H, W, C going through a GAP operation to form a vector. There are connections marked k=5 between two rows of circles, followed by a rectangle and an element-wise multiplication with a circle. It ends with output dimensions H, W, C.</alt-text>
</graphic>
</fig>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mfrac>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>+</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>b</mml:mi>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where the hyperparameters are b and &#x3b3;, the local interactions between channels are then captured by a 1D convolution operation using k, which produces the channel weights <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mi>C</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> using a Sigmoid activation function. Lastly, the improved feature maps F&#x2019; are obtained by multiplying the weights <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msub>
<mml:mi>M</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> by the input feature maps channel by channel.</p>
</sec>
<sec id="s2_2_2_4" sec-type="intro">
<label>2.2.2.4</label>
<title>Introduction of attention mechanisms</title>
<p>The architecture of the ResNet50 network improved with attention mechanisms is shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>. In order to increase its emphasis on significant characteristics and boost classification performance, this model incorporates attention modules at many critical stages compared to the standard ResNet50 framework. The network&#x2019;s first convolutional layer (Conv1) extracts low-level information. Following Conv2 and Conv3, attention modules are added in the convolutional stages (Stage 2). For comparison study, these modules (designated as *Attention*) stand for SENet, CBAM, and ECA, which are separately included in ResNet50. Lastly, the network uses a fully connected layer (FC), a Softmax layer, and global average pooling (AvgPool) to classify. The residual network&#x2019;s performance in complex tasks can be improved by better focusing on important feature areas by incorporating attention processes.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>ResNet50 model with integrated attention mechanism.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g009.tif">
<alt-text content-type="machine-generated">Diagram of a convolutional neural network architecture divided into three stages. Stage 1 includes input, convolution, batch normalization, ReLU activation, and max pooling. Stages 2 and 3 feature sequential convolutional blocks, attention modules, average pooling, a fully connected layer, and softmax output. Each stage reduces dimensions and increases feature depth.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s2_2_3">
<label>2.2.3</label>
<title>Depthwise separable convolution</title>
<p>Additionally, we added depth-separable convolution to the sorghum seed recognition model&#x2019;s later, residual blocks to lower the network model&#x2019;s computational expense and time consumption. As illustrated in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>, the Depthwise Separable Convolution comprises depth and point-by-point convolution (<xref ref-type="bibr" rid="B6">Chollet, 2017</xref>). By breaking down the computational process, depthwise separable convolution drastically lowers the computational cost and parameter count compared to traditional convolution. Conventional convolution, which has a high computational cost, combines channel feature fusion and spatial feature extraction into a single operation. Each convolution kernel acts on all input channels to produce an output channel. Deep separable convolution, on the other hand, divides this process into two steps: the first stage uses deep convolution to extract spatial features by performing the convolution operation independently on each channel, and the second stage uses point-by-point convolution in the channel dimension for weighted combination to realize channel feature fusion. This independent design reduces the amount of computation while maintaining the feature extraction capability of the model, which is an essential component of lightweight models.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Structure of the depthwise separable convolution.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g010.tif">
<alt-text content-type="machine-generated">Diagram illustrating a convolutional neural network with depthwise and pointwise convolutions. It starts with a three-channel input followed by three sets of filters. Depthwise convolution is applied, producing three feature maps. Next, pointwise convolution occurs with four filters, generating four output maps. Lines indicate data flow between elements.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_2_4">
<label>2.2.4</label>
<title>Proposed model</title>
<p>The final process of classifying sorghum seeds by the improved network is shown in <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>. Depth separable convolution (DSC) at various convolutional layers and the CBAM attention mechanism are introduced in this network design to maximize the model performance. Through the weighting mechanism of channel attention and spatial attention, the CBAM module, which is inserted explicitly after the Conv2 and Conv3 convolutional layers, enhances the model&#x2019;s capacity to concentrate on essential features and lessens interference from background noise. Furthermore, by breaking down the standard convolution into depth convolution and point-by-point convolution, depth separable convolution (DSC), which is employed in the Conv4 convolutional layer, not only lowers the computational complexity and number of parameters but also enhances computational efficiency while ensuring the feature extraction capability. This strategy greatly optimizes the model&#x2019;s resource usage while increasing the classification accuracy of sorghum seeds by enabling the network to fuse multi-modal input from RGB images and spectral curves more effectively.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Classification process.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g011.tif">
<alt-text content-type="machine-generated">Diagram of a neural network architecture processing fruit images. It includes layers labeled Convolutional, Max Pooling, Cbam, DSC, Average-Pooling, and Fully Connected-Softmax. Images and graphs flank each side, depicting input and output data.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Overall flow chart</title>
<p>
<xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref> shows a general flow chart. The entire image processing and model-building procedure for sorghum seeds is depicted in this figure. First, an industrial camera and hyperspectral acquisition equipment gather RGB and hyperspectral data from seeds. The appearance features of the RGB images are extracted using a seed segmentation module, and the spectral features of each pixel are extracted from the hyperspectral data using a spectral curve. In order to increase the accuracy of classification and recognition, the RGB and hyperspectral data are fused to create a feature map that blends spectral and spatial information. An attention mechanism and deep separable convolution were added to the model design using the ResNet residual network to further improve the classification and recognition performance of sorghum seeds. In order to accomplish practical and precise seed recognition, the entire approach focuses on multi-modal data fusion and lightweight model construction.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Overall flow chart.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g012.tif">
<alt-text content-type="machine-generated">Flowchart depicting a process involving four main stages: Data Collection, Image-Spectral Fusion, Model Construction, and Model Optimization. Data Collection includes images of apples and a spectral graph. Image-Spectral Fusion combines RGB images and hyperspectral data. Model Construction and Optimization detail neural network architectures with several stages, showing layers like Convolutional, Max Pooling, and Fully Connected. Arrows indicate the workflow between stages.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Indicators for model evaluation</title>
<p>The model&#x2019;s classification performance was thoroughly evaluated in this work using five widely used assessment metrics: accuracy, specificity, recall, Precision, and F1-score. A comprehensive evaluation of the model across various class distributions is provided by the F1-score, which is the reconciled mean of Precision and Recall. Accuracy, on the other hand, reflects the overall correctness of the model&#x2019;s classification; Specificity measures the model&#x2019;s ability to identify negative class samples and emphasizes the importance of reducing false alarms; Recall indicates the model&#x2019;s sensitivity to positive class samples and focuses on lowering underreporting; and Precision is used to assess the model&#x2019;s accuracy in predicting positive classes, reflecting the reliability of the classification results. By properly evaluating the model&#x2019;s strengths and weaknesses locally and overall, combining these metrics enables a thorough examination of the model&#x2019;s performance in the classification task. It serves as a foundation for additional optimization.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Experimental procedures</title>
<p>The following experimental environment was used for this study: Windows 11 as the operating system, the 12th generation Intel<sup>&#xae;</sup> Core&#x2122; i9-12900K (3.20 GHz) as the processor, the NVIDIA GeForce RTX 3090Ti as the graphics configuration, and Pycharm 2021 Community Edition as the integrated development environment. The PyTorch deep learning framework was used to build and train the sorghum seed categorization model. During the model training process, the SGD (Stochastic Gradient Descent) optimizer was selected, with the initial learning rate set to 0.001, and the weight parameters were adjusted to optimize the network loss function. Each epoch represents a complete training cycle for the entire sorghum seed dataset, and its maximum number of rounds was set to 50 in order to obtain the optimal value of the loss function during the training process. In addition, the minimum batch size was set to 8, the momentum parameter was set to 0.9, and the weight decay coefficient was set to 0.01 to enhance the generalization ability of the model and suppress overfitting.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and discussions</title>
<sec id="s3_1">
<label>3.1</label>
<title>Fusion of RGB image data with hyperspectral data</title>
<p>The classification performance of RGB, hyperspectral, and RGB &amp; HSI data on several ResNet models is displayed in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The findings demonstrate that combining RGB and hyperspectral data can significantly enhance classification accuracy. In particular, the ResNet50 model achieves the highest accuracy (0.8961), Precision (0.8969), recall (0.8961), and F1-score (0.8959) with fused data, demonstrating that fused data (RGB &amp; HSI) significantly improves the classification performance in all models. With fused data, other models like ResNet34 also saw notable performance gains. Furthermore, the fused data showed enhanced Specificity and Recall, suggesting that the model could lower the false detection rate and recognize seed classes more accurately by merging RGB and HSI information. Data fusion significantly improved the classification model&#x2019;s performance by combining the complementary nature of spectral and spatial information. ResNet50 performed best in both data cases, indicating that it can be used as the preferred model for classifying sorghum seeds.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Comparison of results before and after fusion.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Data</th>
<th valign="top" align="left">Model</th>
<th valign="top" align="left">Accuracy</th>
<th valign="top" align="left">Specificity</th>
<th valign="top" align="left">Recall</th>
<th valign="top" align="left">Precision</th>
<th valign="top" align="left">F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="4">RGB</td>
<td valign="top" align="left">ResNet18</td>
<td valign="top" align="left">0.7602</td>
<td valign="top" align="left">0.9659</td>
<td valign="top" align="left">0.7607</td>
<td valign="top" align="left">0.7608</td>
<td valign="top" align="left">0.7609</td>
</tr>
<tr>
<td valign="top" align="left">ResNet34</td>
<td valign="top" align="left">0.8055</td>
<td valign="top" align="left">0.9722</td>
<td valign="top" align="left">0.8056</td>
<td valign="top" align="left">0.8055</td>
<td valign="top" align="left">0.8054</td>
</tr>
<tr>
<td valign="top" align="left">ResNet50</td>
<td valign="top" align="left">0.8664</td>
<td valign="top" align="left">0.9817</td>
<td valign="top" align="left">0.8664</td>
<td valign="top" align="left">0.8662</td>
<td valign="top" align="left">0.8650</td>
</tr>
<tr>
<td valign="top" align="left">ResNet101</td>
<td valign="top" align="left">0.7148</td>
<td valign="top" align="left">0.9593</td>
<td valign="top" align="left">0.7148</td>
<td valign="top" align="left">0.7148</td>
<td valign="top" align="left">0.7148</td>
</tr>
<tr>
<td valign="top" align="left" rowspan="4">HSI</td>
<td valign="top" align="left">ResNet18</td>
<td valign="top" align="left">0.7749</td>
<td valign="top" align="left">0.9679</td>
<td valign="top" align="left">0.7750</td>
<td valign="top" align="left">0.7774</td>
<td valign="top" align="left">0.7748</td>
</tr>
<tr>
<td valign="top" align="left">ResNet34</td>
<td valign="top" align="left">0.8438</td>
<td valign="top" align="left">0.9776</td>
<td valign="top" align="left">0.8430</td>
<td valign="top" align="left">0.8441</td>
<td valign="top" align="left">0.8417</td>
</tr>
<tr>
<td valign="top" align="left">ResNet50</td>
<td valign="top" align="left">0.8812</td>
<td valign="top" align="left">0.9833</td>
<td valign="top" align="left">0.8827</td>
<td valign="top" align="left">0.8865</td>
<td valign="top" align="left">0.8825</td>
</tr>
<tr>
<td valign="top" align="left">ResNet101</td>
<td valign="top" align="left">0.7679</td>
<td valign="top" align="left">0.9682</td>
<td valign="top" align="left">0.7754</td>
<td valign="top" align="left">0.7840</td>
<td valign="top" align="left">0.7752</td>
</tr>
<tr>
<td valign="top" align="left" rowspan="4">RGB&amp;HSI</td>
<td valign="top" align="left">ResNet18</td>
<td valign="top" align="left">0.8117</td>
<td valign="top" align="left">0.9758</td>
<td valign="top" align="left">0.8263</td>
<td valign="top" align="left">0.8333</td>
<td valign="top" align="left">0.8260</td>
</tr>
<tr>
<td valign="top" align="left">ResNet34</td>
<td valign="top" align="left">0.8656</td>
<td valign="top" align="left">0.9808</td>
<td valign="top" align="left">0.8656</td>
<td valign="top" align="left">0.8657</td>
<td valign="top" align="left">0.8648</td>
</tr>
<tr>
<td valign="top" align="left">ResNet50</td>
<td valign="top" align="left">0.8961</td>
<td valign="top" align="left">0.9852</td>
<td valign="top" align="left">0.8961</td>
<td valign="top" align="left">0.8969</td>
<td valign="top" align="left">0.8959</td>
</tr>
<tr>
<td valign="top" align="left">ResNet101</td>
<td valign="top" align="left">0.8086</td>
<td valign="top" align="left">0.9727</td>
<td valign="top" align="left">0.8086</td>
<td valign="top" align="left">0.8072</td>
<td valign="top" align="left">0.8072</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We compare the visual confusion matrix for the model before and after fusion in <xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref> to further confirm that the fusion can enhance the model&#x2019;s performance. The confusion matrix data describes the sample&#x2019;s actual categories and the categories that the classifier predicted. Usually, there are four metrics: false positives (FP), false negatives (FN), true positives (TP), and true negatives (TN) (<xref ref-type="bibr" rid="B10">Javanmardi et&#xa0;al., 2021</xref>). The fused models (a2-d2) demonstrated a significant increase in the number of correct classifications on the diagonal and a substantial decrease in misclassifications compared to the unfused models (a-d, a1-d1). This suggests that fusing multi-modal data can improve the models&#x2019; feature differentiation ability.</p>
<fig id="f13" position="float">
<label>Figure&#xa0;13</label>
<caption>
<p>Confusion matrix before and after fusion. Confusion matrix: <bold>(a)</bold> RGB-ResNet18, <bold>(b)</bold> RGB-ResNet34, <bold>(c)</bold> RGB-ResNet50, <bold>(d)</bold> RGB-ResNet101; <bold>(a1)</bold> HSI-ResNet18, <bold>(b1)</bold> HSI-ResNet34, <bold>(c1)</bold> HSI-ResNet50, <bold>(d1)</bold> HSI-ResNet101; (a2) RGB&amp;HSI-ResNet18, (b2) RGB&amp;HSI-ResNet34, <bold>(c2)</bold> RGB&amp;HSI-ResNet50, <bold>(d2)</bold> RGB&amp;HSI-ResNet101.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g013.tif">
<alt-text content-type="machine-generated">Twelve confusion matrices arranged in a grid corresponding to labels (a) to (d2). Each matrix plots true labels against predicted labels from 127 to 187, using a color gradient to indicate prediction accuracy. Each label configuration slightly varies, displaying differences in model performance visualized by intensity in the blue shading. Each matrix's diagonal typically shows higher values, indicating correct predictions, while off-diagonal values represent misclassifications.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_2" sec-type="intro">
<label>3.2</label>
<title>Introduction of an attention mechanism</title>
<p>Three distinct attentional processes are added to the ResNet50 network independently for comparison in order to determine the best network model because of its high performance. After combining RGB and HSI data, <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> shows how adding various attention methods (SE, CBAM, and ECA) affects the ResNet50 model&#x2019;s classification performance. Adding the attention mechanisms enhances the model&#x2019;s overall classification performance, with CBAM-ResNet50 exhibiting the best results. With an accuracy of 93.20%, recall and Precision of 92.63% and 92.71%, respectively, and an F1-score of 92.43%, CBAM-ResNet50 specifically outperforms the others in every category. While the performance of the residual network with the SE module added is marginally worse than that of the CBAM and ECA modules, it is still far better than the model without the attention mechanism included.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Results of RGB&amp;HSI data after introducing the attention mechanism.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Data</th>
<th valign="top" align="left">Model</th>
<th valign="top" align="left">Accuracy</th>
<th valign="top" align="left">Specificity</th>
<th valign="top" align="left">Recall</th>
<th valign="top" align="left">Precision</th>
<th valign="top" align="left">F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" rowspan="3">RGB&amp;HSI</td>
<td valign="top" align="left">SE-ResNet50</td>
<td valign="top" align="left">0.9125</td>
<td valign="top" align="left">0.9886</td>
<td valign="top" align="left">0.9192</td>
<td valign="top" align="left">0.9193</td>
<td valign="top" align="left">0.9190</td>
</tr>
<tr>
<td valign="top" align="left">CBAM-ResNet50</td>
<td valign="top" align="left">0.9320</td>
<td valign="top" align="left">0.9890</td>
<td valign="top" align="left">0.9263</td>
<td valign="top" align="left">0.9271</td>
<td valign="top" align="left">0.9243</td>
</tr>
<tr>
<td valign="top" align="left">ECA-ResNet50</td>
<td valign="top" align="left">0.9203</td>
<td valign="top" align="left">0.9885</td>
<td valign="top" align="left">0.9199</td>
<td valign="top" align="left">0.9238</td>
<td valign="top" align="left">0.9197</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Adding the attention mechanism can significantly increase the model&#x2019;s capacity to identify important features and enhance classification performance, particularly following the fusing of multi-modal data (RGB &amp; HSI). In this experiment, the combination with the best classification effect is CBAM-ResNet50.</p>
<p>The confusion matrix in <xref ref-type="fig" rid="f14">
<bold>Figure&#xa0;14</bold>
</xref> also shows that adding various attention strategies enhances the model&#x2019;s classification performance. The mechanism can effectively focus on the important information in the spatial dimension to improve accuracy, as evidenced by the Resnet50 model with the addition of CBAM having the clearest diagonal of the confusion matrix and a further decrease in misclassifications.</p>
<fig id="f14" position="float">
<label>Figure&#xa0;14</label>
<caption>
<p>Confusion matrix of the result of introducing the attention mechanism after fusion Confusion matrix: <bold>(a)</bold> RGB&amp;HSI-SE-ResNet50, <bold>(b)</bold> RGB&amp;HSI-CBAM-ResNet50, <bold>(c)</bold> RGB&amp;HSI-ECA-ResNet50.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g014.tif">
<alt-text content-type="machine-generated">Three side-by-side confusion matrices labeled (a), (b), and (c). Each matrix compares predicted labels to true labels, ranging from 127 to 187, with values indicating the frequency of predictions. The matrices use a blue color gradient to represent higher counts.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Introducing depth separable convolution</title>
<p>The comparison of classification performance outcomes following the addition of depth separable convolution (DSC) to the RGB &amp; HSI-CBAM-ResNet50 models is shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. The model combines the optimal design of CBAM and DSC based on merging RGB and HSI data to produce good performance in all measures. In particular, the model&#x2019;s accuracy of 94.84% indicates a high level of classification precision overall; its specificity of 99.20% shows that it can effectively lower the false detection rate; and its recall and precision of 94.39% and 94.52%, respectively, show that it can identify every category in real classification. The capacity of actual classification to identify each category is more balanced. Furthermore, the model&#x2019;s outstanding performance in striking a balance between recall and Precision is further confirmed by the F1-score, which reaches 94.38%.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparison of introducing depth-separable convolution.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Data</th>
<th valign="top" align="left">Model</th>
<th valign="top" align="left">Accuracy</th>
<th valign="top" align="left">Specificity</th>
<th valign="top" align="left">Recall</th>
<th valign="top" align="left">Precision</th>
<th valign="top" align="left">F1-score</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">RGB&amp;HSI</td>
<td valign="top" align="left">CBAM-ResNet50</td>
<td valign="top" align="left">0.9320</td>
<td valign="top" align="left">0.9890</td>
<td valign="top" align="left">0.9263</td>
<td valign="top" align="left">0.9271</td>
<td valign="top" align="left">0.9243</td>
</tr>
<tr>
<td valign="top" align="left">RGB&amp;HSI</td>
<td valign="top" align="left">CBAM-ResNet50-DSC</td>
<td valign="top" align="left">0.9484</td>
<td valign="top" align="left">0.9920</td>
<td valign="top" align="left">0.9439</td>
<td valign="top" align="left">0.9452</td>
<td valign="top" align="left">0.9438</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The confusion matrix before and after the implementation of DSC is contrasted in <xref ref-type="fig" rid="f15">
<bold>Figure&#xa0;15</bold>
</xref>. It is evident from this confusion matrix that the model can correctly identify samples of every category because the diagonal in <xref ref-type="fig" rid="f15">
<bold>Figure&#xa0;15b</bold>
</xref> has the great majority of correct classifications. Further evidence is that the CBAM-ResNet50-DSC model has greatly improved in feature extraction, and the critical area of emphasis is the low misclassifications in non-diagonal locations. Furthermore, the confusion matrix&#x2019;s balanced classification performance for many categories demonstrates the model&#x2019;s excellent generalization and stability, which offers a solid foundation for further applications.</p>
<fig id="f15" position="float">
<label>Figure&#xa0;15</label>
<caption>
<p>Comparison of introducing depth-separable convolution. Confusion matrix: <bold>(a)</bold> RGB&amp;HSI-CBAM-ResNet50, <bold>(b)</bold> RGB&amp;HSI-CBAM-ResNet50-DSC.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g015.tif">
<alt-text content-type="machine-generated">Dual confusion matrices labeled (a) and (b) compare predicted versus true labels. Darker blue diagonal indicates correct predictions. Both have varying off-diagonal values representing misclassifications. Colorbar shows scale from 0 to 160.</alt-text>
</graphic>
</fig>
<p>The model maintains effective feature extraction capabilities while drastically lowering the computational cost thanks to deep separable convolution. Higher accuracy and robustness in the classification job are achieved by the model&#x2019;s increased attention to the important feature regions when combined with the channel and spatial attention mechanism of CBAM. The design considerably improves the application value of multi-modal data fusion techniques and offers a superior solution for the sorghum seed classification job.</p>
</sec>
<sec id="s3_4" sec-type="results">
<label>3.4</label>
<title>Results</title>
<p>
<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> shows that, compared to the previous model, the modified model improved several metrics for each of the eight sorghum seed types. Accuracy increased by 8.75%, 0.63%, 3.13%, 30.63%, 3.12%, 5.00%, 8.75%, and 5.62% for every sorghum seed. The corresponding improvements in recall were 8.75%, 0.62%, 3.12%, 27.12%, 3.13%, 5.00%, 6.91%, and 5.62%. The corresponding improvements in Precision were 10.02%, 5.39%, 11.49%, 12.79%, 3.13%, 12.13%, 0.10%, and 14.06%. Furthermore, there was a 0.0941, 0.0305, 0.0747, 0.2031, 0.0313, 0.0856, 0.0013, and 0.0963 improvement in the F1 scores, respectively. These findings suggest that the enhanced network performs better in recognition when categorizing maize seeds&#x2019; photos.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Performance comparison of single-species models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" rowspan="2" align="center">Seed category</th>
<th valign="top" colspan="2" align="center">Accuracy</th>
<th valign="top" colspan="2" align="center">Recall</th>
<th valign="top" colspan="2" align="center">Precision</th>
<th valign="top" colspan="2" align="center">F1-score</th>
</tr>
<tr>
<th valign="top" align="center">Before</th>
<th valign="top" align="center">After</th>
<th valign="top" align="center">Before</th>
<th valign="top" align="center">After</th>
<th valign="top" align="center">Before</th>
<th valign="top" align="center">After</th>
<th valign="top" align="center">Before</th>
<th valign="top" align="center">After</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">127</td>
<td valign="top" align="center">0.8500</td>
<td valign="top" align="center">0.9375</td>
<td valign="top" align="center">0.8500</td>
<td valign="top" align="center">0.9375</td>
<td valign="top" align="center">0.8144</td>
<td valign="top" align="center">0.9146</td>
<td valign="top" align="center">0.8318</td>
<td valign="top" align="center">0.9259</td>
</tr>
<tr>
<td valign="top" align="center">136</td>
<td valign="top" align="center">0.9875</td>
<td valign="top" align="center">0.9938</td>
<td valign="top" align="center">0.9875</td>
<td valign="top" align="center">0.9937</td>
<td valign="top" align="center">0.9461</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9664</td>
<td valign="top" align="center">0.9969</td>
</tr>
<tr>
<td valign="top" align="center">141</td>
<td valign="top" align="center">0.9625</td>
<td valign="top" align="center">0.9938</td>
<td valign="top" align="center">0.9625</td>
<td valign="top" align="center">0.9937</td>
<td valign="top" align="center">0.8851</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9222</td>
<td valign="top" align="center">0.9969</td>
</tr>
<tr>
<td valign="top" align="center">159</td>
<td valign="top" align="center">0.6625</td>
<td valign="top" align="center">0.9688</td>
<td valign="top" align="center">0.6625</td>
<td valign="top" align="center">0.9337</td>
<td valign="top" align="center">0.7681</td>
<td valign="top" align="center">0.8960</td>
<td valign="top" align="center">0.7114</td>
<td valign="top" align="center">0.9145</td>
</tr>
<tr>
<td valign="top" align="center">160</td>
<td valign="top" align="center">0.9688</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9687</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9687</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9687</td>
<td valign="top" align="center">1.0000</td>
</tr>
<tr>
<td valign="top" align="center">177</td>
<td valign="top" align="center">0.7875</td>
<td valign="top" align="center">0.8375</td>
<td valign="top" align="center">0.7875</td>
<td valign="top" align="center">0.8375</td>
<td valign="top" align="center">0.7545</td>
<td valign="top" align="center">0.8758</td>
<td valign="top" align="center">0.7706</td>
<td valign="top" align="center">0.8562</td>
</tr>
<tr>
<td valign="top" align="center">186</td>
<td valign="top" align="center">0.9125</td>
<td valign="top" align="center">1.0000</td>
<td valign="top" align="center">0.9125</td>
<td valign="top" align="center">0.9816</td>
<td valign="top" align="center">0.9799</td>
<td valign="top" align="center">0.9800</td>
<td valign="top" align="center">0.9450</td>
<td valign="top" align="center">0.9463</td>
</tr>
<tr>
<td valign="top" align="center">187</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.8562</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.8562</td>
<td valign="top" align="center">0.8108</td>
<td valign="top" align="center">0.9514</td>
<td valign="top" align="center">0.8050</td>
<td valign="top" align="center">0.9013</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>As can be observed from the radargram in <xref ref-type="fig" rid="f16">
<bold>Figure&#xa0;16a</bold>
</xref>, the RGB&amp;HSI-CBAM-ResNet50-DSC model has the largest overall encompassing area and the strongest sorghum seed recognition ability. It also achieves the best results in all five performance metrics: Accuracy, Specificity, Recall, Precision, and F1-score. The RGB&amp;HSI-CBAM-ResNet50 model, on the other hand, came in second, suggesting that the DSC module&#x2019;s addition improved the model&#x2019;s performance even further. The RGB&amp;HSI-ResNet50 model outperformed the RGB-ResNet50 or HSI-ResNet50 models individually, suggesting that adding the attention mechanism and combining multi-modal information enhanced the model recognition effect. Regarding metrics like recall and F1-score, RGB-ResNet50 performs the lowest among them, indicating its limitations when tackling hyperspectral fine classification jobs. Therefore, multi-modal fusion and model optimization are crucial for improving model robustness and recognition.</p>
<fig id="f16" position="float">
<label>Figure&#xa0;16</label>
<caption>
<p>
<bold>(a)</bold> Radar chart <bold>(b)</bold> Comparison of model losses <bold>(c)</bold> Comparison of model accuracy.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g016.tif">
<alt-text content-type="machine-generated">(a) Radar chart comparing models on accuracy, specificity, recall, precision, F1-score. (b) Line graph showing model loss over 50 epochs for five models with varying loss values. (c) Line graph depicting model accuracy over epochs for five models, with accuracy improving over time.</alt-text>
</graphic>
</fig>
<p>The validation data loss for each epoch is displayed in <xref ref-type="fig" rid="f16">
<bold>Figure&#xa0;16b</bold>
</xref>. The loss value comparison graphs show that all five models&#x2019; loss values gradually decrease as the number of training rounds increases. This indicates that the models gradually converge during the training process. In the model performance comparison experiments, we compared the accuracy and loss values of the five models with the number of training rounds. With the lowest loss value, the RGB&amp;HSI-CBAM-ResNet50-DSC model exhibits superior optimization and more effective training.</p>
<p>The accuracy of the validation data for every epoch is displayed in <xref ref-type="fig" rid="f16">
<bold>Figure&#xa0;16c</bold>
</xref>. The RGB&amp;HSI-CBAM-ResNet50-DSC model attains the highest accuracy of over 90% at the late stage of training. Still, the accuracy of all the models rises with the number of training rounds, as seen in the accuracy comparison graph. This finding implies that the model&#x2019;s classification performance can be considerably enhanced by integrating RGB and HSI information and adding CBAM and DSC methods. It suggests that enhanced network structure and greater feature information are necessary for the model to perform well on the job.</p>
<p>In conclusion, the findings demonstrate that the RGB&amp;HSI-CBAM-ResNet50-DSC model achieves the best accuracy and loss values, confirming the usefulness of enhancing the model&#x2019;s structure and combining multi-modal features.</p>
<p>To comprehensively evaluate the improvement in model performance and its statistical robustness, each model in this study was independently trained and tested ten times to ensure result stability and reproducibility. Based on these repeated experiments, a systematic statistical analysis was conducted using 95% confidence intervals and one-way analysis of variance (ANOVA) to assess classification performance across different models. The results indicate that with the gradual integration of hyperspectral information, multi-modal fusion strategies, the attention mechanism (CBAM), and depthwise separable convolution (DSC), the model accuracy steadily increased from 0.86596 for the baseline RGB-ResNet50 model (95% CI: [0.86513, 0.86679]) to 0.94779 for the final RGB&amp;HSI-CBAM-ResNet50-DSC model (95% CI: [0.94727, 0.94831]). Furthermore, the narrowing of the confidence intervals indicates not only higher accuracy but also improved performance stability.</p>
<p>The subsequent ANOVA analysis revealed that the performance differences among the models were statistically highly significant (with p-values far below 0.05) across all five core evaluation metrics: Accuracy, F1-score, Recall, Precision, and Specificity. Additionally, three key indicators&#x2014;Accuracy, F1-score, and Recall&#x2014;were visualized using boxplots, as shown in <xref ref-type="fig" rid="f17">
<bold>Figure&#xa0;17</bold>
</xref>. These plots intuitively illustrate the median values and interquartile ranges of each model&#x2019;s performance on different metrics. The results further confirm the trend of consistent performance improvement as the model architecture is gradually optimized. The final model demonstrated the best and most stable performance across all metrics. In summary, the combined statistical analysis and visualizations validate the effectiveness, robustness, and practical value of the proposed multi-modal fusion and structural optimization strategies in the classification task of sorghum seeds.</p>
<fig id="f17" position="float">
<label>Figure&#xa0;17</label>
<caption>
<p>Boxplots of accuracy, F1-score, and recall for different models.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1632698-g017.tif">
<alt-text content-type="machine-generated">Boxplot comparing accuracy, F1-score, and recall for different models: RGB-ResNet50, HSI-ResNet50, RGB&amp;HSI-ResNet50, RGB&amp;HSI-CBAM-ResNet50, and RGB&amp;HSI-CBAM-ResNet50-DSC. Box colors differentiate models, with scores ranging from 0.86 to 0.94.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Ablation experiments</title>
<p>Using ResNet50 as the backbone, we conducted a series of ablation experiments to assess the impact of spectral fusion, attention mechanisms, and depthwise separable convolution (DSC) on model performance. The incorporation of the Convolutional Block Attention Module (CBAM) significantly improved classification accuracy by enhancing the model&#x2019;s focus on critical spatial and channel-wise features. However, this enhancement came at the cost of increased parameter count and computational complexity, resulting in a slight rise in inference time. To mitigate this, DSC was introduced to replace standard convolutions, substantially reducing both the model size and computational load. Specifically, the number of parameters was reduced from 26.9 million to 17.4 million, while FLOPs decreased from 4.31G to 2.62G. When CBAM and DSC were combined, the model achieved the best trade-off between performance and efficiency-reaching the highest classification accuracy with only 18.3 million parameters and 2.92G FLOPs. Remarkably, the inference time remained comparable to or even slightly lower than the baseline model, highlighting the excellent potential of the proposed architecture for deployment in resource-constrained agricultural environments, as shown in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Comparison of ResNet50 experimental models with different module combinations.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">ResNet50</th>
<th valign="top" align="center">Fusion</th>
<th valign="top" align="center">cbam</th>
<th valign="top" align="center">dsc</th>
<th valign="top" align="center">Accuracy</th>
<th valign="top" align="center">Params (M)</th>
<th valign="top" align="center">FLOPs (G)</th>
<th valign="top" align="center">Time (s)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">0.8664</td>
<td valign="top" align="center">25.6</td>
<td valign="top" align="center">4.10</td>
<td valign="top" align="center">0.0024</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">0.8961</td>
<td valign="top" align="center">25.6</td>
<td valign="top" align="center">4.10</td>
<td valign="top" align="center">0.0024</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">0.9107</td>
<td valign="top" align="center">26.9</td>
<td valign="top" align="center">4.31</td>
<td valign="top" align="center">0.0027</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">0.9024</td>
<td valign="top" align="center">17.4</td>
<td valign="top" align="center">2.62</td>
<td valign="top" align="center">0.0020</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">0.9375</td>
<td valign="top" align="center">26.9</td>
<td valign="top" align="center">4.31</td>
<td valign="top" align="center">0.0027</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">0.9389</td>
<td valign="top" align="center">17.4</td>
<td valign="top" align="center">2.62</td>
<td valign="top" align="center">0.0020</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">0.9196</td>
<td valign="top" align="center">18.3</td>
<td valign="top" align="center">2.92</td>
<td valign="top" align="center">0.0023</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">0.9484</td>
<td valign="top" align="center">18.3</td>
<td valign="top" align="center">2.92</td>
<td valign="top" align="center">0.0023</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>&#x2713; Indicates the presence of the corresponding module in the model. This setup is used to facilitate comparative analysis in the ablation study.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>Validation of model generalization ability</title>
<p>To further assess the generalization ability and robustness of the proposed model, three publicly available seed image datasets from the Kaggle platform were selected for validation. As most public datasets contain only RGB images and lack corresponding hyperspectral data, only RGB-based evaluation was conducted in this section. The selected datasets include seven rice seed varieties, three maize seed varieties, and five soybean seed varieties. For each dataset, the performance of the proposed CBAM-ResNet50-DSC model was compared with two widely used baseline models, VGG16 and DenseNet121.As shown in <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>, the proposed CBAM-ResNet50-DSC consistently achieved the highest classification accuracy across all three datasets-89.62% for rice seeds, 89.69% for maize seeds, and 94.73% for soybean seeds-outperforming both VGG16 and DenseNet121. These results demonstrate that the proposed model not only exhibits strong robustness and cross-dataset generalization but also adapts well to different categories of seed samples. Furthermore, the model shows broad applicability and practical value for real-world scenarios involving heterogeneous RGB image data from diverse sources.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Sources and classification accuracy of different public seed datasets.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Data Source Link</th>
<th valign="top" align="center">Model</th>
<th valign="top" align="center">Accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center" rowspan="3">
<ext-link ext-link-type="uri" xlink:href="https://data.mendeley.com/datasets/v6vzvfszj6">https://data.mendeley.com/datasets/v6vzvfszj6</ext-link>
</td>
<td valign="top" align="center">VGG16</td>
<td valign="top" align="center">0.8452</td>
</tr>
<tr>
<td valign="top" align="center">DenseNet121</td>
<td valign="top" align="center">0.8562</td>
</tr>
<tr>
<td valign="top" align="center">CBAM-ResNet50-DSC</td>
<td valign="top" align="center">0.8962</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="3">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.34740/kaggle/dsv/8681789">https://doi.org/10.34740/kaggle/dsv/8681789</ext-link>
</td>
<td valign="top" align="center">VGG16</td>
<td valign="top" align="center">0.8576</td>
</tr>
<tr>
<td valign="top" align="center">DenseNet121</td>
<td valign="top" align="center">0.8745</td>
</tr>
<tr>
<td valign="top" align="center">CBAM-ResNet50-DSC</td>
<td valign="top" align="center">0.8969</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="3">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.34740/kaggle/dsv/6457847">https://doi.org/10.34740/kaggle/dsv/6457847</ext-link>
</td>
<td valign="top" align="center">VGG16</td>
<td valign="top" align="center">0.8838</td>
</tr>
<tr>
<td valign="top" align="center">DenseNet121</td>
<td valign="top" align="center">0.9174</td>
</tr>
<tr>
<td valign="top" align="center">CBAM-ResNet50-DSC</td>
<td valign="top" align="center">0.9473</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusions</title>
<p>This study proposes a sorghum seed variety classification approach based on RGB and hyperspectral (HSI) data fusion and an enhanced deep residual convolutional network (ResNet) for quick, nondestructive, and highly accurate seed identification. ResNet50 was utilized as the base network for model optimization by combining image data fusion with spectral data, the attention-based mechanism (CBAM), and the deep separable convolution (DSC). The spectral fusion dataset included 12,800 seeds from eight different varieties of sorghum seeds.</p>
<p>The experiment&#x2019;s findings indicate that:</p>
<list list-type="order">
<list-item>
<p>Multi-modal data fusion (RGB&amp;HSI) can improve the model&#x2019;s classification performance. The classification accuracy of the fused data is increased to 89.61%, and the F1-score is improved to 0.8959 compared with single data.</p>
</list-item>
<list-item>
<p>The model&#x2019;s capacity to collect important features is further improved by adding the CBAM attention mechanism, raising the classification accuracy to 93.75%&#x2014;4.14% higher than the base ResNet50.</p>
</list-item>
<list-item>
<p>The model&#x2019;s computational efficiency is further maximized by combining it with depth separable convolution (DSC). In addition to lowering the number of model parameters and computational complexity, introducing DSC based on the CBAM-ResNet50 structure improves the final classification accuracy to 94.84%, a 1.09% improvement over the CBAM-ResNet50 model. This confirms the efficacy of the lightweight design.</p>
</list-item>
</list>
<p>Data fusion, attention mechanism optimization, and network lightweight design were used in this study to build an accurate and efficient sorghum seed classification model successfully. This model has high agricultural application value and offers a scientific foundation for variety screening and seed quality detection.</p>
<p>Although the proposed CBAM-ResNet50-DSC model based on 2D spectrogram fusion demonstrates excellent classification accuracy and robustness under experimental conditions, several challenges remain for real-world agricultural applications. First, the quality of RGB and hyperspectral images collected in field environments may be significantly affected by uncontrolled factors such as lighting variations, seed placement angles, image focus, and surface contamination (e.g., dust or aging). These factors can lead to instability in model predictions, thereby limiting practical effectiveness. Second, although this study simplifies the multi-modal fusion process at the model level, acquiring both RGB and hyperspectral images in practice still requires additional imaging equipment. This introduces extra hardware costs and operational complexity, which may hinder large-scale deployment in real field scenarios.</p>
<p>To address these issues, future research will focus on robustness enhancement strategies, domain adaptation techniques, and more cost-effective imaging solutions (e.g., compact multispectral sensors). Furthermore, incorporating multi-temporal or multi-angle data may further improve feature consistency and model stability, thus promoting the practical application of this method in precision agriculture.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>XY: Supervision, Writing &#x2013; review &amp; editing, Funding acquisition, Resources. YC: Validation, Data curation, Conceptualization, Software, Investigation, Writing &#x2013; original draft. SS: Writing &#x2013; review &amp; editing, Methodology, Supervision, Data curation. ZZ: Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. The Natural Science Foundation Project of Jilin Provincial Department of Science and Technology: &#x201c;Research on Phenotypic Information Extraction and Identification Model of Staple Crop Seeds Based on Polarization Imaging Technology&#x201d; (Project No.20250102046JC).</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>An</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>W.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Tensor based low rank representation of hyperspectral images for wheat seeds varieties identification</article-title>. <source>Comput. Electrical Eng.</source> <volume>110</volume>, <fpage>108890</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compeleceng.2023.108890</pub-id>
</citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andersson</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yahya</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Johansson</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Liew</surname> <given-names>J</given-names>
</name>
</person-group>. (<year>2007</year>). <article-title>Seed desiccation tolerance and storage behaviour in Cordeauxia edulis</article-title>. <source>Seed Sci. Technol.</source> <volume>35</volume>, <fpage>660</fpage>&#x2013;<lpage>673</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.15258/sst.2007.35.3.13</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Asmussen</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Conrad</surname> <given-names>O.</given-names>
</name>
<name>
<surname>G&#xfc;nther</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kirsch</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Riller</surname> <given-names>U.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Semi-automatic segmentation of petrographic thin section images using a seeded-region growing algorithm with an application to characterize wheathered subarkose sandstone</article-title>. <source>Comput. Geosciences</source> <volume>83</volume>, <fpage>89</fpage>&#x2013;<lpage>99</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cageo.2015.05.001</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bi</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Bi</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Non-destructive classification of maize seeds based on RGB and hyperspectral data with improved Grey Wolf optimization algorithms</article-title>. <source>Agronomy</source> <volume>14</volume>, <fpage>645</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agronomy14040645</pub-id>
</citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L.-T.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>A.-Q.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>L.-L.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>X.-L.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>M.-L.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>Seed vigor evaluation based on adversity resistance index of wheat seed germination under stress conditions</article-title>. <source>Ying Yong Sheng tai xue bao= J. Appl. Ecol.</source> <volume>27</volume>, <fpage>2968</fpage>&#x2013;<lpage>2974</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.13287/j.1001-9332.201609.014</pub-id>, PMID: <pub-id pub-id-type="pmid">29732861</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chollet</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Xception: Deep learning with depthwise separable convolutions</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>, <publisher-loc>Honolulu, HI, USA</publisher-loc>, <publisher-name>IEEE</publisher-name>, <page-range>1251&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2017.195</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Franco</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Marulanda</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Cruz</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Morales</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Fuentes</surname> <given-names>L. S.</given-names>
</name>
<name>
<surname>Rubiano</surname> <given-names>V.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). &#x201c;<article-title>A neural network approach to predicting viability of native seeds from their optic RGB images</article-title>,&#x201d; in <conf-name>2020 IEEE Symposium Series on Computational Intelligence (SSCI)</conf-name>, <publisher-loc>Honolulu, HI, USA</publisher-loc>, <publisher-name>IEEE</publisher-name>, <page-range>921&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/SSCI47803.2020.9308252</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Cotton leaf disease detection method based on improved SSD</article-title>. <source>Int. J. Agric. Biol. Eng.</source> <volume>17</volume>, <fpage>211</fpage>&#x2013;<lpage>220</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.25165/j.ijabe.20241702.8574</pub-id>
</citation></ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition</conf-name>, <publisher-loc>Las Vegas, NV, USA</publisher-loc>, <publisher-name>IEEE</publisher-name>, <page-range>770&#x2013;8</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Javanmardi</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Ashtiani</surname> <given-names>S. H. M.</given-names>
</name>
<name>
<surname>Verbeek</surname> <given-names>F. J.</given-names>
</name>
<name>
<surname>Martynenko</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Computer-vision classification of corn seed varieties using deep convolutional neural network</article-title>. <source>J. Stored Products Res.</source> <volume>92</volume>, <fpage>101800</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jspr.2021.101800</pub-id>
</citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>An attention mechanism-improved YOLOv7 object detection algorithm for hemp duck count estimation</article-title>. <source>Agriculture</source> <volume>12</volume>, <fpage>1659</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/agriculture12101659</pub-id>
</citation></ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamruzzaman</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Makino</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Oshita</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Rapid and non-destructive&#xa0;detection of chicken adulteration in minced beef using visible near-infrared hyperspectral imaging and machine learning</article-title>. <source>J. Food Eng.</source> <volume>170</volume>, <fpage>8</fpage>&#x2013;<lpage>15</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jfoodeng.2015.08.023</pub-id>
</citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khoddami</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Messina</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Vadabalija Venkata</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Farahnaky</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Blanchard</surname> <given-names>C. L.</given-names>
</name>
<name>
<surname>Roberts</surname> <given-names>T. H.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Sorghum in foods: Functionality and potential in innovative products</article-title>. <source>Crit. Rev. Food Sci. Nutr.</source> <volume>63</volume>, <fpage>1170</fpage>&#x2013;<lpage>1186</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/10408398.2021.1960793</pub-id>, PMID: <pub-id pub-id-type="pmid">34357823</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cui</surname> <given-names>W.-G.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>Y.-Z.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Z.-Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Epileptic seizure detection in EEG signals using a unified temporal-spectral squeeze-and-excitation network</article-title>. <source>IEEE Trans. Neural Syst. Rehabil. Eng.</source> <volume>28</volume>, <fpage>782</fpage>&#x2013;<lpage>794</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TNSRE.7333</pub-id>, PMID: <pub-id pub-id-type="pmid">32078551</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Malik</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Ram</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Arumugam</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Predicting gypsum tofu quality from soybean seeds using hyperspectral imaging and machine learning</article-title>. <source>Food Control</source> <volume>160</volume>, <fpage>110357</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.foodcont.2024.110357</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Masuda</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Suzuki</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Baba</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Takeshita</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Suzuki</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Sugiura</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Noninvasive diagnosis of seedless fruit using deep learning in persimmon</article-title>. <source>Horticulture J.</source> <volume>90</volume>, <fpage>172</fpage>&#x2013;<lpage>180</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2503/hortj.UTD-248</pub-id>
</citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>SPECIAL: zero-shot hyperspectral image classification with CLIP</article-title>. <source>arXiv preprint arXiv:2501.16222</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2501.16222</pub-id>
</citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shahin</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Symons</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Poysa</surname> <given-names>V. W.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Determining soya bean seed size uniformity with image analysis</article-title>. <source>Biosyst. Eng.</source> <volume>94</volume>, <fpage>191</fpage>&#x2013;<lpage>198</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2006.02.011</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Soares</surname> <given-names>S. F. C.</given-names>
</name>
<name>
<surname>Medeiros</surname> <given-names>M. D.</given-names>
</name>
<name>
<surname>Silva</surname> <given-names>R. T. da</given-names>
</name>
<name>
<surname>Santos</surname> <given-names>C. C.</given-names>
</name>
<name>
<surname>Pereira</surname> <given-names>A. F. G.</given-names>
</name>
<name>
<surname>Oliveira</surname> <given-names>P. D.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <article-title>Classification of individual cotton seeds with respect to variety using near-infrared hyperspectral imaging</article-title>. <source>Analytical Methods</source> <volume>8</volume>, <fpage>8498</fpage>&#x2013;<lpage>8505</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1039/C6AY02896A</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sunil</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Koparan</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Ahmed</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Howatt</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Weed and crop species classification using computer vision and deep learning technologies in greenhouse conditions</article-title>. <source>J. Agric. Food Res.</source> <volume>9</volume>, <fpage>100325</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jafr.2022.100325</pub-id>
</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Torchio</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Giacosa</surname> <given-names>S.</given-names>
</name>
<name>
<surname>R&#xed;o Segade</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Gerbi</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Rolle</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Berry heterogeneity as a possible factor affecting the potential of seed mechanical properties to classify wine grape varieties and estimate flavanol release in wine-like solution</article-title>. <source>South Afr. J. Enology Viticulture</source> <volume>35</volume>, <fpage>20</fpage>&#x2013;<lpage>42</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.21548/35-1-982</pub-id>
</citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Traore</surname> <given-names>B. B.</given-names>
</name>
<name>
<surname>Kamsu-Foguem</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Tangara</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Tiako</surname> <given-names>P. F.</given-names>
</name>
<name>
<surname>Foguem</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Da Silva</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Deep convolution neural network for image recognition</article-title>. <source>Ecol. Inf.</source> <volume>48</volume>, <fpage>257</fpage>&#x2013;<lpage>268</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2018.10.002</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tyagi</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Duraisamy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Radhakrishnan</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Suresh</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>A. P.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Non-destructive method for assessing fruit quality using modified depthwise separable convolutions on hyperspectral images</article-title>. <source>Sens. Agric. Food Qual. Saf. XVI SPIE</source> <volume>12670</volume>, <fpage>126700J</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1117/12.3015521</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zuo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). &#x201c;<article-title>ECA-net: efficient channel attention for deep convolutional neural networks</article-title>,&#x201d; in <conf-name>2020 IEEE/CVF conference on computer vision and pattern recognition (CVPR), IEEE</conf-name>, <page-range>11531&#x2013;39</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01155</pub-id>
</citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Modeling and performance evaluation of wind turbine based on ant colony optimization-extreme learning machine</article-title>. <source>Appl. Soft Computing</source> <volume>94</volume>, <fpage>106476</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.asoc.2020.106476</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wenchao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhi</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Research on strawberry disease diagnosis based on improved residual network recognition model</article-title>. <source>Math. Problems Eng.</source> <volume>2022</volume>, <fpage>6431942</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2022/6431942</pub-id>
</citation></ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Woo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J. Y.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Cbam: Convolutional&#xa0;block&#xa0;attention module</article-title>,&#x201d; in <conf-name>Proceedings of the European conference&#xa0;on&#xa0;computer vision (ECCV)</conf-name> <publisher-name>Springer</publisher-name>, <publisher-loc>Cham</publisher-loc>, <page-range>3&#x2013;19</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-01234-2_1</pub-id>
</citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ghamisi</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Chanussot</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Spectralmamba: Efficient mamba for hyperspectral image classification</article-title>. <source>arXiv preprint arXiv:2404.08489</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2404.08489</pub-id>
</citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Bean</surname> <given-names>S. R.</given-names>
</name>
<name>
<surname>Tuinstra</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Seib</surname> <given-names>P. A.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>X. S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>).&#xa0;<article-title>Evaluation of the multi-seeded (msd) mutant of sorghum for ethanol production</article-title>. <source>Ind. Crops products</source> <volume>97</volume>, <fpage>345</fpage>&#x2013;<lpage>353</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.indcrop.2016.12.015</pub-id>
</citation></ref>
</ref-list>
</back>
</article>