<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2025.1604382</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Intelligent recognition of tobacco leaves states during curing with deep neural network</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Xu</surname>
<given-names>Qiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2693397/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Yanling</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Aiguo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Guangqing</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cai</surname>
<given-names>Xianjie</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhou</surname>
<given-names>Shuoye</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Junying</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jin</surname>
<given-names>Baofeng</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yan</surname>
<given-names>Ding</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Huang</surname>
<given-names>Jiajie</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Zuxiao</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Heng</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Jianwei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Guo</surname>
<given-names>Weimin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Jianjun</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Tobacco Agricultral Labaratory, Zhengzhou Tobacco Research Institute of China National Tobacco Corporation (CNTC)</institution>, <addr-line>Zhengzhou</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Tobacco Leaf Administration Office, Henan Provincial Tobacco Company of CNTC</institution>, <addr-line>Zhengzhou</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Tobacco Raw Materials Procurement Center, Shanghai Tobacco Group Co. Ltd.</institution>, <addr-line>Shanghai</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Pingdingshan Branch, Henan Provincial Tobacco Company</institution>, <addr-line>Pingdingshan</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Technical Center, China Tobacco Guangdong Industrial Co. Ltd</institution>, <addr-line>Guangzhou</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Technical Center, Jilin Tobacco Industry Co. Ltd.</institution>, <addr-line>Changchun</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Xing Yang, Anhui Science and Technology University, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Chao Kang, Guizhou University, China</p>
<p>Budi Yanto, Universitas Pasir Pangaraian, Indonesia</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Weimin Guo, <email xlink:href="mailto:guoweimin1984@sina.com">guoweimin1984@sina.com</email>; Jianjun Liu, <email xlink:href="mailto:liujj325@163.com">liujj325@163.com</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1604382</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>06</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Xu, Zhang, Wang, Chen, Cai, Zhou, Li, Jin, Yan, Huang, Chen, Zhang, Wang, Guo and Liu</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Xu, Zhang, Wang, Chen, Cai, Zhou, Li, Jin, Yan, Huang, Chen, Zhang, Wang, Guo and Liu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>The state monitoring of tobacco leaves during the curing process is crucial for process control and automation of tobacco agricultural production. While most of the existing research on tobacco leaves state recognition focused on the temporal state of the leaves, the morphological state was often neglected. Moreover, the previous research typically used a limited number of non-industrial images for training, creating a significant disparity with the images encountered in actual applications.</p>
</sec>
<sec>
<title>Methods</title>
<p>To investigate the potential of deep learning algorithms in identifying the morphological states of tobacco leaves in real industrial scenarios, a comprehensive and large-scale dataset was developed in this study. This dataset focused on the states of tobacco leaves in actual bulk curing barn in multiple production areas in China, specifically recognizing the degrees of yellowing, browning, and drying. Then, an efficient deep learning method was proposed based on this dataset to enhance the predictive performance.</p>
</sec>
<sec>
<title>Results</title>
<p>The prediction accuracy achieved for the yellowing degree, browning degree, and drying degree were 83.0%, 90.5%, and 75.6% respectively. The overall average accuracy, satisfied the requirements of practical application scenarios with a value of 83%.</p>
</sec>
<sec>
<title>Discussion</title>
<p>Our proposed framework effectively enables morphological state recognition in industrial curing, supporting parameter optimization and enhanced tobacco quality.</p>
</sec>
</abstract>
<kwd-group>
<kwd>tobacco leaves</kwd>
<kwd>large-scale dataset</kwd>
<kwd>bulk curing barn</kwd>
<kwd>image recognition</kwd>
<kwd>deep learning</kwd>
</kwd-group>
<contract-num rid="cn001">110202201051 (SJ-01), 110202101084 (SJ-08)</contract-num>
<contract-sponsor id="cn001">China National Tobacco Corporation<named-content content-type="fundref-id">10.13039/501100008862</named-content>
</contract-sponsor>
<counts>
<fig-count count="9"/>
<table-count count="9"/>
<equation-count count="6"/>
<ref-count count="26"/>
<page-count count="12"/>
<word-count count="5306"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Sustainable and Intelligent Phytoprotection</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Curing is an important process of tobacco production, converting fresh leaves into commercially cigarette raw materials. Curing quality of tobacco leaves directly determines farmers&#x2019; income and the cigarette quality. Matching the optimal curing technology in real time according to the state of tobacco leaves is the key to determine the curing quality of tobacco leaves (<xref ref-type="bibr" rid="B15">Siddiqui, 2001</xref>; <xref ref-type="bibr" rid="B26">Zhao et&#xa0;al., 2024</xref>). At present, the identification of tobacco leaf states during the curing process mainly relies on people&#x2019;s subjective experience. Inaccurate cognition has caused problems such as uneven curing quality of tobacco leaves, large curing losses, and weak industrial usability (<xref ref-type="bibr" rid="B12">Ma et&#xa0;al., 2021</xref>).</p>
<p>The rapid development of technologies such as the Internet of Things and artificial intelligence has proposed new methods for solving such problems. Researchers conducted research on the state of tobacco leaves during the curing process by using the collected temperature, humidity and tobacco leaf images. The researches on tobacco leaf state identification mainly can be divided into two categories: temporal state and morphological state. The temporal state of tobacco leaves refers to the division of the tobacco leaf curing process into different stages based on the curing time of the tobacco leaves, such as yellowing stage, color fixing stage, and stem drying stage (<xref ref-type="bibr" rid="B8">Li et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B11">Lu et&#xa0;al., 2023</xref>). Although this method has achieved high accuracy (More than 90%), it is difficult to adjust the temperature and humidity of the curing room in real time based on the recognition results. Therefore, some researchers focused on the morphological state recognition (<xref ref-type="bibr" rid="B18">Wang et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B19">Wang and Qin, 2022</xref>; <xref ref-type="bibr" rid="B26">Zhao et&#xa0;al., 2024</xref>). The morphological state of tobacco leaves refers to the specific state of yellowing degree, drying degree and browning degree of tobacco leaves identified based on tobacco leaf images, thereby replacing the human eye observation and subjective analysis during the curing process, and providing more accurate, faster and scientific results for identifying the state of tobacco leaves. Meanwhile, it can also provide an important reference for the real-time adjustment of the curing technology (<xref ref-type="bibr" rid="B2">Condor&#xed; et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B13">Pei et&#xa0;al., 2024</xref>). To further improve the recognition accuracy, the texture information of tobacco leaf images has also begun to be gradually utilized except the widely used color information (<xref ref-type="bibr" rid="B18">Wang et&#xa0;al., 2017</xref>).</p>
<p>However, there are still some problems limited the accuracy and application of these recognition models. Previous research on tobacco leaf state recognition often relied on small-scale (i.e., hundreds to just over a thousand samples) (<xref ref-type="bibr" rid="B26">Zhao et&#xa0;al., 2024</xref>) or non-industrial datasets, which were collected using small ovens, experimental chambers, etc (<xref ref-type="bibr" rid="B21">Wu and Yang, 2021</xref>; <xref ref-type="bibr" rid="B2">Condor&#xed; et&#xa0;al., 2020</xref>). For example, some studies have acquired images through smartphone photography (<xref ref-type="bibr" rid="B7">Howard et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B25">Zhang et&#xa0;al., 2023</xref>), but the quality and characteristics of these images differ significantly from those captured in actual curing barns, limiting their applicability to real-world bulk curing scenarios. Some researchers tried to collect tobacco images in the actual curing barns, but the complex environment during curing process resulted in image distortion, out of focus, obvious color difference and only partial tobacco image acquisition, which are still the core problems limiting the acquisition of tobacco condition information (<xref ref-type="bibr" rid="B2">Condor&#xed; et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B13">Pei et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B22">Wu et&#xa0;al., 2014</xref>; <xref ref-type="bibr" rid="B20">Wu and Yang, 2019</xref>; <xref ref-type="bibr" rid="B24">Zhang et&#xa0;al., 2013</xref>) and further effected the wide application of the recognition models.</p>
<p>To overcome these limitations, the objective of this study is to (i) construct a comprehensive and large-scale image dataset captured directly from actual bulk curing barns; (ii),propose a deep learning approach to recognize the morphological states of tobacco leaves throughout the curing process based on this dataset; (iii) establish a benchmark framework using state-of-the-art models, including the Swin Transformer V2, to enhance predictive performance and support intelligent decision-making during tobacco curing.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Large-scale curing tobacco leaves dataset</title>
<p>The large-scale curing tobacco leaves dataset involved gathering a substantial amount of real-world data from bulk curing barn and having them meticulously labeled by experts in tobacco curing. To facilitate the recognition of tobacco leaves states during the curing process, 17,420 images of tobacco leaves from 10 main production areas in China were collected, including Henan, Fujian, Yunnan, Guizhou, etc. All tobacco leaf images in the dataset were collected by a newly developed autonomous imaging device (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). The image device was installed in the middle shed on one side of the grill near the heating chamber in the curing barns (<xref ref-type="bibr" rid="B23">Xu et&#xa0;al., 2024</xref>). The sampling interval was set to 10 minutes, and the tobacco images of the curing process were obtained, which marked the time, location, temperature, and humidity in the curing barns and the status of the tobacco leaves.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Image acquisition device and installation photos of tobacco leaves during curing process.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g001.tif">
<alt-text content-type="machine-generated">A shelf is mounted on a wall with visible support brackets. Below the shelf, a small electronic device is attached with wires hanging down. The wall is composed of vertical panels, and there is a narrow window near the ceiling.</alt-text>
</graphic>
</fig>
<p>Experts in tobacco curing in China conducted evaluations focusing on three distinct states of the tobacco leaf: the degree of yellowing, the degree of browning, and the degree of drying. Different degrees were categorized based on the extent of morphological differences observed in the various states of the tobacco leaves (<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>). In <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, some reference images along were provided with their corresponding yellowing degree, browning degree, and drying degree labels for further clarity. This visual representation aids in understanding the various states and degrees of tobacco leaves during the curing process.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>The detailed definition of the states of tobacco leaf.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">States</th>
<th valign="top" align="center">Degrees</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="7" align="left">Yellowing degree</td>
<td valign="top" align="left">0. 50%&#x223c;60% yellowing</td>
</tr>
<tr>
<td valign="top" align="left">1. 70%&#x223c;80% yellowing</td>
</tr>
<tr>
<td valign="top" align="left">2. Leaves yellow, veins green and green base</td>
</tr>
<tr>
<td valign="top" align="left">3. Leaves yellow and veins green</td>
</tr>
<tr>
<td valign="top" align="left">4. Main veins fade cyan to white</td>
</tr>
<tr>
<td valign="top" align="left">5. Partial main veins shrink and turn purple</td>
</tr>
<tr>
<td valign="top" align="left">6. Main vein purpling</td>
</tr>
<tr>
<td valign="middle" rowspan="9" align="left">Drying degree</td>
<td valign="top" align="left">0. Leaf swell and harden</td>
</tr>
<tr>
<td valign="top" align="left">1. Leaf tip softening</td>
</tr>
<tr>
<td valign="top" align="left">2. Leaf softening</td>
</tr>
<tr>
<td valign="top" align="left">3. Leaf wilt completely</td>
</tr>
<tr>
<td valign="top" align="left">4. Leaf blade hook tip curl</td>
</tr>
<tr>
<td valign="top" align="left">5. Leaf drying 1/2&#x223c;2/3</td>
</tr>
<tr>
<td valign="top" align="left">6. Leaf drying completely (Large roll)</td>
</tr>
<tr>
<td valign="top" align="left">7. Main vein drying 1/2</td>
</tr>
<tr>
<td valign="top" align="left">8. Main vein drying completely</td>
</tr>
<tr>
<td valign="middle" rowspan="6" align="left">Browning degree</td>
<td valign="top" align="left">0. 0%</td>
</tr>
<tr>
<td valign="top" align="left">1.&lt;10%</td>
</tr>
<tr>
<td valign="top" align="left">2. 10%&#x223c;20%</td>
</tr>
<tr>
<td valign="top" align="left">3. 20%&#x223c;30%</td>
</tr>
<tr>
<td valign="top" align="left">4. 30%&#x223c;50%</td>
</tr>
<tr>
<td valign="top" align="left">5. <italic>&gt;</italic>50%</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Reference image data of <bold>(A)</bold> Pingdingshan and <bold>(B)</bold> Zunyi of different yellowing and drying degrees of tobacco leaves during curing.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g002.tif">
<alt-text content-type="machine-generated">Two sets of images labeled A and B depict stages of leaf yellowing and drying. Both series show similar phases: initial yellowing, leaf softening, vein color changes, and eventual drying. Text details the gradual process, indicating percentages of yellowing and specific leaf changes like swelling, wilting, vein fading, and purpling through their progressive stages.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Recognition algorithm</title>
<sec id="s2_2_1">
<label>2.2.1</label>
<title>Method overview</title>
<p>A recognition algorithm was implemented to accurately and efficiently identify tobacco states during the curing process based on deep neural networks. As depicted in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, three components were comprised in the recognition algorithm: (i) a pre-trained backbone pre-trained on universal image recognition datasets (e.g., ImageNet), (ii) a Fourier filter module, and (iii) a common color filter module. The pre-trained backbone extracted highly discriminative image features <italic>F<sub>img</sub>
</italic> by leveraging previously learned information and further fine-tunes the parameters on the proposed datasets. The Fourier filter module was designed to extract the wrinkle information of tobacco leaves <italic>F<sub>spect</sub>
</italic> by utilizing the Fourier spectrum map of the image and a convolution-based network. The common color filter module calculated the quantized color histogram of the image and filtered it by frequency, thereby screening out high-frequency colors and the order of the colors appearing in the image. It further employed a fully connected layer to extract high-frequency color features <italic>F<sub>color</sub>
</italic>. Finally, all features were concatenated and fed into the fully connected classifier to predict three states of tobacco.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>The overview of the recognition algorithm, including three key components: <bold>(A)</bold> a pre-trained backbone (Swin-Transformer v2), <bold>(B)</bold> a Common Color Filter Module and <bold>(C)</bold> a Fourier Filter Module. Each module was designed to extract distinct features, which were then integrated and fed into the subsequent classification network to predict the states of tobacco leaves.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g003.tif">
<alt-text content-type="machine-generated">Diagram showing a process for analyzing tobacco images. The image undergoes a pre-trained backbone (A), common color filter module (B) using color histograms, and a Fourier filter module (C) using convolutional layers and Fourier transform. Outputs are connected to a classifier determining browning, yellowing, and drying degrees.</alt-text>
</graphic>
</fig>
<p>In summary, a three-branch network was presented in this study incorporating two proposed modules: the Fourier filter module and the common color filter module. Through joint training, the network employs a multi-label recognition head to simultaneously classify three tobacco states.</p>
</sec>
<sec id="s2_2_2">
<label>2.2.2</label>
<title>Pre-trained backbone</title>
<p>To build a deep learning benchmark and verify the efficacy of deep learning networks for the tobacco leaves states recognition, four extensively employed deep neural networks pre-trained on ImageNet were utilized as backbone network, including VGG19, ResNet-152, ViT, Swin-Transformer and Swin-Transformer v2 (<xref ref-type="bibr" rid="B16">Simonyan and Zisserman, 2014</xref>; <xref ref-type="bibr" rid="B5">He et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B4">Dosovitskiy et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B10">Liu et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B9">2022</xref>). VGG19 is a profound convolutional neural network comprising 16 convolution layers and 3 fully connected layers. ResNet-152 is an exceptionally deep convolutional neural network, reaching a depth of up to 152 layers, made possible by employing skip connections to bypass certain layers. ViT-Large is a model that applies the transformer architecture, which has demonstrated impressive performance in the field of computer vision recently. The Swin Transformer is a hierarchical vision model engineered for efficient image recognition. It utilizes non-overlapping windows and self-attention within each window to process images at multiple scales. Swin-Transformer V2 enhances this approach with innovations like scaled cosine attention, post- normalization, and a log-spaced continuous position bias, boosting stability, scalability, and overall performance.</p>
<p>These pre-trained backbones extract highly discriminative image features by leveraging the information learned before and further fine-tuning the parameters on the proposed datasets. An evaluation of the predictive accuracy of these four networks in determining the state of tobacco was conducted. To further enhance their performance, Swin-Transformer v2 was incorporated as the backbone network and its core components including the following two aspects.</p>
<sec id="s2_2_2_1">
<label>2.2.2.1</label>
<title>The attention mechanism</title>
<p>In the Swin-Transformer v2, the attention mechanism is a crucial component (<xref ref-type="bibr" rid="B17">Vaswani et&#xa0;al., 2017</xref>). It performs multiple attention operations to extract highly discriminative features. Given N image patches within an image and their corresponding features F &#x2208; RN&#xd7;d&#x2032; (the process to obtain F will be detailed in the subsequent paragraph), the operation of the self-attention mechanism is as follows:</p>
<p>First, the query (Q), key (K), and value (V) are computed using a linear transformation with the trainable weight W &#x2208; R<sup>d&#x2032;&#xd7;3d</sup> (<xref ref-type="disp-formula" rid="eq1">Equations 1</xref>):</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>V</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>W</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where Q, K, V &#x2208; R<sup>N&#xd7;d</sup>, W represents the weight matrix of a linear layer, where <italic>d<sup>&#x2032;</sup>
</italic> is the input channel dimension of the feature F &#x2208; K V &#xd7; <italic>d<sup>&#x2032;</sup>
</italic>, <italic>d</italic> is the output channel dimension of the linear layer. Then, the attention mechanism is applied to extract the output feature (<xref ref-type="disp-formula" rid="eq2">Equations 2</xref>, <xref ref-type="disp-formula" rid="eq3">3</xref>):</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>K</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>X</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:msubsup>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where cos(Q, K) &#x2208; R<sup>N&#xd7;N</sup> is the pair-wise cosine similarity, &#x3b3; is a learnable scalar and <italic>N</italic> denotes the number of image patches. The B &#x2208; R<sup>N&#xd7;N</sup> serves as a relative positional encoding which is predicted by two trainable fully connected layers g (<xref ref-type="disp-formula" rid="eq4">Equation 4</xref>):</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x394;</mml:mi>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x394;</mml:mi>
<mml:mi>X</mml:mi>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>&#x394;</mml:mi>
<mml:mi>Y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
</sec>
<sec id="s2_2_2_2">
<label>2.2.2.2</label>
<title>Patch splitting and merging</title>
<p>The attention mechanism is implemented within the image patches, which are derived from segmenting an input image into non-overlapping patches. Each patch is regarded as an individual unit, referred to as a &#x201c;token&#x201d;. Then a linear projection transforms each token to the token features F. The token features are subsequently fed into numerous layers of the Swin-Transformer V2. Each layer is designed to extract and refine the information embedded within the tokens. This refinement process involves a series of operations that integrate attention mechanisms. As the tokens progress further into the depths of the network, a method known as patch merging is employed. This method reduces the number of tokens by integrating the features of neighboring patches. The result is an ensemble of tokens, thereby effectively establishing a hierarchical representation of the initial image.</p>
</sec>
</sec>
<sec id="s2_2_3">
<label>2.2.3</label>
<title>Fourier filter module</title>
<p>The prediction of drying is more dependent on the morphological characteristics of the tobacco leaves compared to the prediction of yellowing and browning. As depicted in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, images of tobacco leaves with a higher degree of drying display a greater number of wrinkles, which are associated with the high-frequency information in the image&#x2019;s Fourier spectrum. As the drying process progresses, the formation of surface wrinkles on tobacco leaves increases, thereby amplifying the high-frequency intensity in the corresponding images.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Images of tobacco leaves with <bold>(A)</bold> low drying and <bold>(B)</bold> high drying degrees.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g004.tif">
<alt-text content-type="machine-generated">Panel A shows a close-up of yellow and green leaves, possibly drying, with visible temperature and humidity data. Panel B displays a similar arrangement of leaves, indicating a progression in drying, also with environmental data.</alt-text>
</graphic>
</fig>
<p>To validate this hypothesis, the average high-frequency and low-frequency intensities of images with different degrees of drying within the training set were computed. The results, as illustrated in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, revealed that the average high-frequency intensity of the corresponding image exhibits an upward trend as the degree of drying increases. This suggested a correlation between the image&#x2019;s frequency domain information and its degree of drying. Consequently, the Fourier spectrum was incorporated as information into the network and a Fourier filter was construct to aid in the prediction of the degrees of drying.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Statistical (normalized) high/low frequency intensity of images with different drying degrees.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g005.tif">
<alt-text content-type="machine-generated">Line graph showing frequency intensity against dryness degree. The blue line represents low frequency and decreases initially, then stabilizes. The green line represents high frequency and increases steadily, surpassing the blue line around degree three.</alt-text>
</graphic>
</fig>
<p>Specifically, the image was converted into a gray-scale image at first and then its two-dimensional Fourier spectrum was calculated (<xref ref-type="disp-formula" rid="eq5">Equation 5</xref>):</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>H</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>v</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">/</mml:mo>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where F is the Fourier spectrum with the same shape as the input image. The H and W are the height and width, respectively. Then the real and imaginary components of each frequency position within this matrix are utilized as input for the neural network. Where the values of real and imaginary components are represented as separate image channels. Four-layer convolutional layers are constructed to extract spectral features Fspect from this input. This approach enables the effective capture of intricate patterns within the Fourier spectrum, thereby enhancing the robustness of the drying prediction.</p>
</sec>
<sec id="s2_2_4">
<label>2.2.4</label>
<title>Common color filter module</title>
<p>In conjunction with the Fourier Filter module, which is primarily designed to augment the prediction of drying, an additional module denoted as the Common Color Filter was introduced. This module was specifically engineered to enhance the prediction accuracy of the states intrinsically tied to the color of the tobacco leaves. However, the image signal frequently encompasses elements beyond the mere color of the tobacco, the presence of noise color could potentially compromise the final prediction. Therefore, it is important to eliminate as many noisy pixels as possible to mitigate color interference. To address this challenge, the characteristic that the tobacco in the bulk curing barn is densely arranged and typically occupies a consistent position were exploited. This strategy aided in the effective reduction of noise and enhanced the accuracy of the proposed model.</p>
<p>As depicted in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>, this algorithm filtered out the most common colors in an image and extracts relevant features for subsequent use. It accomplished this through a series of steps. Firstly, it applied a center-cropping technique to the image. This process focused on the central part of the image, which contained the most important information and reduced the impact of potential noise from the image&#x2019;s periphery. Next, it quantized the color space. Quantization was a process that reduced the number of distinct colors used in an image, while still maintaining its overall visual construction. This step can reduce the size of the color space and the computation load when calculating the color histogram. Following this, the most common colors were selected. These colors are shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>, it depicted typically the tobacco leaves. By focusing on these colors, the algorithm can more accurately predict the state of yellowing and browning. Finally, the common color feature Fcolor was extracted using a single fully-connected layer.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Implementation procedure of filtering algorithm.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g006.tif">
<alt-text content-type="machine-generated">Algorithm for a common color filter is detailed. It includes initializing quantization color range, center cropping the image, mapping RGB values, using a statistical histogram, selecting top colors, and extracting features with convolutional neural networks. Parameters include quantization parameter, number of colors reserved, and quantization color range. Steps involve image cropping, RGB mapping, histogram creation, top color selection, and feature extraction.</alt-text>
</graphic>
</fig>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Visualization of filtering process: <bold>(A)</bold> input image, <bold>(B)</bold> quantized and center- cropped image and <bold>(C)</bold> after histogram filtering. The pixel positions corresponding to the retained colors (shown in purple) primarily focus on the tobacco leaves rather than other background areas.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g007.tif">
<alt-text content-type="machine-generated">Three-step image processing sequence on leafy plants. (A) Original image with yellow-green leaves. (B) Image after crop and quantization, focusing on leaf details. (C) Image after color filtering, highlighting areas in pink.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Preprocessing and evaluation metrics</title>
<p>During training, input images were resized to 384&#xd7;384, the supported input size of our image backbone. Data augmentation was applied using the RandAugment method with a magnitude of 9 and a standard deviation of 0.5.</p>
<p>The primary evaluation metric employed to assess the algorithms&#x2019; performance was top-1 accuracy, expressed as a percentage. Specifically, the network&#x2019;s prediction probability for the most likely state was selected as the predicted result and compared with the ground-truth label. The accuracy is then calculated using the (<xref ref-type="disp-formula" rid="eq6">Equation 6</xref>):</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>#</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>#</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>+</mml:mo>
<mml:mo>#</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>#TruePostive</italic> represents instances where the model accurately predicts the positive class, <italic>#FalsePostive</italic> denotes instances of incorrect predictions by the model, the symbol &#x201c;#&#x201d; indicates the number of corresponding instances or categories. The individual prediction accuracy for three different tobacco leaves states as well as their average accuracy were separately evaluated.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and discussion</title>
<sec id="s3_1">
<label>3.1</label>
<title>Accuracy of tobacco leaves state prediction</title>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Comparison with traditional algorithms</title>
<p>The performance of three traditional algorithms on the same task, including K nearest neighbor (KNN), support vector machines (SVM), and random forest (RF), was compared with our deep-learning method (<xref ref-type="bibr" rid="B3">Cover and Hart, 1967</xref>; <xref ref-type="bibr" rid="B6">Hearst et&#xa0;al., 1998</xref>; <xref ref-type="bibr" rid="B1">Breiman, 2001</xref>). Three distinct predicted states of tobacco leaves and hyperparameters for each method are shown in <xref ref-type="table" rid="T2">
<bold>Tables&#xa0;2</bold>
</xref>, <xref ref-type="table" rid="T3">
<bold>3</bold>
</xref>, respectively.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>The prediction accuracy of different traditional methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" colspan="2" align="center">Methods</th>
<th valign="middle" colspan="4" align="center">Accuracy</th>
</tr>
<tr>
<th valign="middle" align="center">Yellowing</th>
<th valign="middle" align="center">Browning</th>
<th valign="middle" align="center">Drying</th>
<th valign="middle" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="3" align="center">Traditional</td>
<td valign="middle" align="center">KNN</td>
<td valign="middle" align="center">66.5</td>
<td valign="middle" align="center">84.5</td>
<td valign="middle" align="center">54.6</td>
<td valign="middle" align="center">68.5</td>
</tr>
<tr>
<td valign="middle" align="center">RF</td>
<td valign="middle" align="center">71.2</td>
<td valign="middle" align="center">84.3</td>
<td valign="middle" align="center">64.6</td>
<td valign="middle" align="center">73.4</td>
</tr>
<tr>
<td valign="middle" align="center">SVM</td>
<td valign="middle" align="center">73.9</td>
<td valign="middle" align="center">85.7</td>
<td valign="middle" align="center">64.4</td>
<td valign="middle" align="center">74.7</td>
</tr>
<tr>
<td valign="middle" align="center">Deep learning</td>
<td valign="middle" align="center">Ours</td>
<td valign="middle" align="center">
<bold>83.0</bold>
</td>
<td valign="middle" align="center">
<bold>90.5</bold>
</td>
<td valign="middle" align="center">
<bold>75.6</bold>
</td>
<td valign="middle" align="center">
<bold>83.0</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold font is the best result in each experiment.</p>
</table-wrap-foot> </table-wrap>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>The hyperparameters of different traditional methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Methods</th>
<th valign="middle" align="center">Hyperparameters</th>
<th valign="middle" align="center">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="2" align="center">KNN</td>
<td valign="middle" align="center">neighbors</td>
<td valign="middle" align="center">5</td>
</tr>
<tr>
<td valign="middle" align="center">distance type</td>
<td valign="middle" align="center">Frobenius norm</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="center">SVM</td>
<td valign="middle" align="center">C</td>
<td valign="middle" align="center">1</td>
</tr>
<tr>
<td valign="middle" align="center">kernel</td>
<td valign="middle" align="center">Linear</td>
</tr>
<tr>
<td valign="middle" align="center">Penalty term</td>
<td valign="middle" align="center">L2</td>
</tr>
<tr>
<td valign="middle" align="center">loss</td>
<td valign="middle" align="center">squared hinge</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="center">RF</td>
<td valign="middle" align="center">Number of trees</td>
<td valign="middle" align="center">100</td>
</tr>
<tr>
<td valign="middle" align="center">Split criterion</td>
<td valign="middle" align="center">Gini impurity</td>
</tr>
<tr>
<td valign="middle" align="center">bootstrap</td>
<td valign="middle" align="center">True</td>
</tr>
<tr>
<td valign="middle" align="center">n_bins</td>
<td valign="middle" align="center">1024</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Firstly, the deep-learning method significantly outperformed the three traditional methods, highlighting the potential benefits of using deep learning for this task. Secondly, the KNN method performed well in predicting browning degree by directly computing the difference between two images as the norm, and was ineffective for yellowing degree and drying degree. This suggested that the prediction of browning degree relied more on the color information within the image.</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Comparison with deep learning algorithms</title>
<p>As shown in <xref ref-type="table" rid="T4">
<bold>Tables&#xa0;4</bold>
</xref>, <xref ref-type="table" rid="T5">
<bold>5</bold>
</xref>, Swin-Transformer v2-Large achieved state-of-the-art performance among all other single backbones and benefited from a larger in-put size. The method proposed in this study further enhanced Swin-Transformer v2-Large&#x2019;s performance, demonstrating a higher accuracy with an average accuracy of 83.0%. To verify the effectiveness of the two proposed modules (FFM and CCFM), The new method was also implemented based on the Swin-Transformer-Large as the pre-trained backbone network. It can be observed that after integrating FFM and CCFM modules with the Swin-Transformer, an improvement in accuracy was achieved (81.8% vs. 81.6%). This indicated that these modules possessed a certain degree of robustness.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>The prediction accuracy of different deep learning methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Methods</th>
<th valign="middle" rowspan="2" align="center">Resolution</th>
<th valign="middle" rowspan="2" align="center">Pre-trained</th>
<th valign="middle" colspan="4" align="center">Accuracy</th>
</tr>
<tr>
<th valign="middle" align="center">Yellowing</th>
<th valign="middle" align="center">Browning</th>
<th valign="middle" align="center">Drying</th>
<th valign="middle" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">VGG-19</td>
<td valign="middle" align="center">224&#xd7;224</td>
<td valign="middle" align="center">ImageNet1K</td>
<td valign="middle" align="center">79.2</td>
<td valign="middle" align="center">87.7</td>
<td valign="middle" align="center">70.6</td>
<td valign="middle" align="center">79.1</td>
</tr>
<tr>
<td valign="middle" align="center">ResNet-152</td>
<td valign="middle" align="center">224&#xd7;224</td>
<td valign="middle" align="center">ImageNet1K</td>
<td valign="middle" align="center">78.7</td>
<td valign="middle" align="center">88.7</td>
<td valign="middle" align="center">70.1</td>
<td valign="middle" align="center">79.2</td>
</tr>
<tr>
<td valign="middle" align="center">ViT-Large</td>
<td valign="middle" align="center">224&#xd7;224</td>
<td valign="middle" align="center">ImageNet22K</td>
<td valign="middle" align="center">80.7</td>
<td valign="middle" align="center">90.1</td>
<td valign="middle" align="center">71.4</td>
<td valign="middle" align="center">80.7</td>
</tr>
<tr>
<td valign="middle" align="center">Swin-Base</td>
<td valign="middle" align="center">224&#xd7;224</td>
<td valign="middle" align="center">ImageNet1K</td>
<td valign="middle" align="center">80.8</td>
<td valign="middle" align="center">89.3</td>
<td valign="middle" align="center">73.0</td>
<td valign="middle" align="center">81.0</td>
</tr>
<tr>
<td valign="middle" align="center">ConvNeXt V2</td>
<td valign="middle" align="center">224&#xd7;224</td>
<td valign="middle" align="center">ImageNet22K</td>
<td valign="middle" align="center">81.5</td>
<td valign="middle" align="center">90.1</td>
<td valign="middle" align="center">74.0</td>
<td valign="middle" align="center">81.9</td>
</tr>
<tr>
<td valign="middle" align="center">Swin-Base</td>
<td valign="middle" align="center">224&#xd7;224</td>
<td valign="middle" align="center">ImageNet22K</td>
<td valign="middle" align="center">81.1</td>
<td valign="middle" align="center">89.7</td>
<td valign="middle" align="center">73.3</td>
<td valign="middle" align="center">81.4</td>
</tr>
<tr>
<td valign="middle" align="center">Swin-Large</td>
<td valign="middle" align="center">224&#xd7;224</td>
<td valign="middle" align="center">ImageNet22K</td>
<td valign="middle" align="center">81.3</td>
<td valign="middle" align="center">90.2</td>
<td valign="middle" align="center">73.2</td>
<td valign="middle" align="center">81.6</td>
</tr>
<tr>
<td valign="middle" align="center">Ours(Swin-Large)</td>
<td valign="middle" align="center">224&#xd7;224</td>
<td valign="middle" align="center">ImageNet22K</td>
<td valign="middle" align="center">81.3</td>
<td valign="middle" align="center">90.3</td>
<td valign="middle" align="center">73.9</td>
<td valign="middle" align="center">81.8</td>
</tr>
<tr>
<td valign="middle" align="center">SwinV2-Large</td>
<td valign="middle" align="center">256&#xd7;256</td>
<td valign="middle" align="center">ImageNet22K</td>
<td valign="middle" align="center">82.2</td>
<td valign="middle" align="center">90.2</td>
<td valign="middle" align="center">73.9</td>
<td valign="middle" align="center">82.1</td>
</tr>
<tr>
<td valign="middle" align="center">SwinV2-Large</td>
<td valign="middle" align="center">384&#xd7;384</td>
<td valign="middle" align="center">ImageNet22K</td>
<td valign="middle" align="center">82.6</td>
<td valign="middle" align="center">89.6</td>
<td valign="middle" align="center">75.0</td>
<td valign="middle" align="center">82.4</td>
</tr>
<tr>
<td valign="middle" align="center">Ours(SwinV2-Large)</td>
<td valign="middle" align="center">384&#xd7;384</td>
<td valign="middle" align="center">ImageNet22K</td>
<td valign="middle" align="center">
<bold>83.0</bold>
</td>
<td valign="middle" align="center">
<bold>90.5</bold>
</td>
<td valign="middle" align="center">
<bold>75.6</bold>
</td>
<td valign="middle" align="center">
<bold>83.0</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold font is the best result in each experiment.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>The hyperparameters of our methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Hyperparameters</th>
<th valign="middle" align="center">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">warmup learning rate</td>
<td valign="middle" align="center">2e-8</td>
</tr>
<tr>
<td valign="middle" align="center">base learning rate</td>
<td valign="middle" align="center">2e-4</td>
</tr>
<tr>
<td valign="middle" align="center">end learning rate</td>
<td valign="middle" align="center">2e-7</td>
</tr>
<tr>
<td valign="middle" align="center">batch size</td>
<td valign="middle" align="center">8</td>
</tr>
<tr>
<td valign="middle" align="center">decay-epoch</td>
<td valign="middle" align="center">5</td>
</tr>
<tr>
<td valign="middle" align="center">learning rate schedule</td>
<td valign="middle" align="center">cosine</td>
</tr>
<tr>
<td valign="middle" align="center">optimizer</td>
<td valign="middle" align="center">AdamW</td>
</tr>
<tr>
<td valign="middle" align="center">drop-path</td>
<td valign="middle" align="center">0.1</td>
</tr>
<tr>
<td valign="middle" align="center">gradient clip</td>
<td valign="middle" align="center">1</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>To further verify the impact of image resolution and pre-trained image datasets on the accuracy of tobacco condition recognition, experiments with different parameters based on Swin-Transformer and Swin-Transformer v2 were conducted. The results indicated that the accuracy of tobacco leaves states recognition can benefit from being pre-trained on a larger image dataset (i.e., ImageNet-22k), even if it was not directly related to tobacco leaves. Higher input resolution can also improve prediction accuracy, suggesting that it is possible to further enhance the accuracy of tobacco leaf recognition by increasing the input resolution. Moreover, this method required low computational cost and provided fast output, with a less than 2GB of GPU memory for inference during testing and an average prediction time of under 0.5 seconds per image.</p>
</sec>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Ablation experiment of the proposed component</title>
<p>Ablation experiments were conducted to demonstrate the effectiveness of the three proposed components in this paper. As depicted in <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>, when compared to the standalone backbone model, the Fourier Filter Module (FFM) contributed an absolute improvement of 1.0% in drying prediction accuracy, underscoring its efficacy in enhancing drying prediction. Similarly, the Common Color Filter Module (CCFM) accounted for an absolute increase of 0.8% in browning prediction, affirming its utility in tasks significantly influenced by color attributes. Moreover, as indicated in the final row, the synergistic integration of both modules lead to further enhancements in prediction precision, thereby confirming that their combined application can substantially bolster overall performance.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>The ablation of the proposed component.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Backbone</th>
<th valign="middle" rowspan="2" align="center">FFM</th>
<th valign="middle" rowspan="2" align="center">CCFM</th>
<th valign="middle" colspan="4" align="center">Accuracy</th>
</tr>
<tr>
<th valign="middle" align="center">Yellowing</th>
<th valign="middle" align="center">Browning</th>
<th valign="middle" align="center">Drying</th>
<th valign="middle" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center"/>
<td valign="top" align="left">82.6</td>
<td valign="top" align="left">89.6</td>
<td valign="top" align="left">75.0</td>
<td valign="top" align="left">82.4</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="top" align="left">83.0</td>
<td valign="top" align="left">89.5</td>
<td valign="top" align="left">76.0</td>
<td valign="top" align="left">82.8</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center"/>
<td valign="middle" align="center">&#x2713;</td>
<td valign="top" align="left">82.5</td>
<td valign="top" align="left">90.4</td>
<td valign="top" align="left">75.4</td>
<td valign="top" align="left">82.8</td>
</tr>
<tr>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="middle" align="center">&#x2713;</td>
<td valign="top" align="left">83.0</td>
<td valign="top" align="left">90.5</td>
<td valign="top" align="left">75.6</td>
<td valign="top" align="left">83.0</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Effectiveness of the hyper-parameters in the common color filter module</title>
<p>The effectiveness of the two hyper-parameters, quantization parameter Q and the number of colors reserved K, related to the common color filter module were evaluated. Different values of Q and K were chosen. The experimental results are shown in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>. The optimum value was obtained when Q = 4 and K = 20. When Q = 2, the color granularity became smaller and showed higher performance in predicting browning and yellowing, and the drying slightly decreased. One possible reason was that this fine color information dominated the feature extraction. Under the same Q, sampling with different K will also lead to different results, indicating that the balance the situations of insufficient sampling and excessive noise sampling through K was required.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>The ablation of the hyper-parameters in the common color filter module.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Q</th>
<th valign="middle" rowspan="2" align="center">K</th>
<th valign="middle" colspan="4" align="center">Accuracy</th>
</tr>
<tr>
<th valign="middle" align="center">Yellowing</th>
<th valign="middle" align="center">Browning</th>
<th valign="middle" align="center">Drying</th>
<th valign="middle" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">10</td>
<td valign="top" align="left">82.7</td>
<td valign="top" align="left">89.8</td>
<td valign="top" align="left">75.3</td>
<td valign="top" align="left">82.6</td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">20</td>
<td valign="top" align="left">83.0</td>
<td valign="top" align="left">
<bold>90.5</bold>
</td>
<td valign="top" align="left">
<bold>75.6</bold>
</td>
<td valign="top" align="left">
<bold>83.0</bold>
</td>
</tr>
<tr>
<td valign="top" align="left">4</td>
<td valign="top" align="left">30</td>
<td valign="top" align="left">82.7</td>
<td valign="top" align="left">90.1</td>
<td valign="top" align="left">75.3</td>
<td valign="top" align="left">82.7</td>
</tr>
<tr>
<td valign="top" align="left">8</td>
<td valign="top" align="left">20</td>
<td valign="top" align="left">83.1</td>
<td valign="top" align="left">90.2</td>
<td valign="top" align="left">74.9</td>
<td valign="top" align="left">82.8</td>
</tr>
<tr>
<td valign="top" align="left">2</td>
<td valign="top" align="left">20</td>
<td valign="top" align="left">
<bold>83.2</bold>
</td>
<td valign="top" align="left">
<bold>90.5</bold>
</td>
<td valign="top" align="left">75.0</td>
<td valign="top" align="left">82.9</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>Bold font is the best result in each experiment.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Effectiveness of production area independent prediction</title>
<p>The large scale dataset comprised images acquired across various production areas, sensor discrepancies, tobacco varieties and ecological environmental variations can all contribute to inherent difference of images. While a unified training approach was straightforward and widely adopted, the impact of area specificity on the recognition of tobacco states was also explored in this study. Three representative production areas with substantial data volume were selected and individual training processes were conducted. In addition, the integrated training model in all areas was used to make separate predictions for these areas and compared them with the results of the separately trained model. This allowed assessments on area-specific models&#x2019; effectiveness and their practicality for different tobacco production areas.</p>
<p>The experiment was conducted in three distinct production areas, labeled as A, B, and C. Each bulk curing barn employed an image device, with the corresponding data statistics presented in <xref ref-type="table" rid="T8">
<bold>Table&#xa0;8</bold>
</xref>. As shown in <xref ref-type="table" rid="T9">
<bold>Table&#xa0;9</bold>
</xref>, it was evident across three distinct areas that the integrated model outperformed models trained individually for each area in terms of prediction accuracy. This suggested that despite the inherent difference present in images from different areas, the model can still leverage a larger image dataset to enhance its performance, surpassing that of models trained individually in each area.</p>
<table-wrap id="T8" position="float">
<label>Table&#xa0;8</label>
<caption>
<p>Data statistics of three typical production areas.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Production area</th>
<th valign="top" align="center">A</th>
<th valign="top" align="center">B</th>
<th valign="top" align="center">C</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Training data</td>
<td valign="top" align="center">1.2k</td>
<td valign="top" align="center">2.9k</td>
<td valign="top" align="center">1.2k</td>
</tr>
<tr>
<td valign="top" align="center">Test data</td>
<td valign="top" align="center">0.5k</td>
<td valign="top" align="center">1.2k</td>
<td valign="top" align="center">0.5k</td>
</tr>
<tr>
<td valign="top" align="center">All data</td>
<td valign="top" align="center">1.7k</td>
<td valign="top" align="center">4.1k</td>
<td valign="top" align="center">1.7k</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T9" position="float">
<label>Table&#xa0;9</label>
<caption>
<p>The comparison of training of different areas separately.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="center">Area</th>
<th valign="middle" colspan="4" align="center">Individual model</th>
<th valign="middle" colspan="4" align="center">Integrated model</th>
</tr>
<tr>
<th valign="middle" align="center">Yellowing</th>
<th valign="middle" align="center">Browning</th>
<th valign="middle" align="center">Drying</th>
<th valign="middle" align="center">All</th>
<th valign="middle" align="center">Yellowing</th>
<th valign="middle" align="center">Browning</th>
<th valign="middle" align="center">Drying</th>
<th valign="middle" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">All</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">83.0</td>
<td valign="top" align="center">90.5</td>
<td valign="top" align="center">75.6</td>
<td valign="top" align="center">83.0</td>
</tr>
<tr>
<td valign="top" align="center">A</td>
<td valign="top" align="center">84.7</td>
<td valign="top" align="center">87.1</td>
<td valign="top" align="center">76.9</td>
<td valign="top" align="center">82.9</td>
<td valign="top" align="center">84.8</td>
<td valign="top" align="center">87.6</td>
<td valign="top" align="center">78.7</td>
<td valign="top" align="center">83.7</td>
</tr>
<tr>
<td valign="top" align="center">B</td>
<td valign="top" align="center">82.2</td>
<td valign="top" align="center">90.9</td>
<td valign="top" align="center">77.7</td>
<td valign="top" align="center">83.6</td>
<td valign="top" align="center">83.7</td>
<td valign="top" align="center">90.6</td>
<td valign="top" align="center">77.5</td>
<td valign="top" align="center">84.0</td>
</tr>
<tr>
<td valign="top" align="center">C</td>
<td valign="top" align="center">86.0</td>
<td valign="top" align="center">89.0</td>
<td valign="top" align="center">77.8</td>
<td valign="top" align="center">84.3</td>
<td valign="top" align="center">84.6</td>
<td valign="top" align="center">90.5</td>
<td valign="top" align="center">79.5</td>
<td valign="top" align="center">84.9</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Visualization of the confusion matrix</title>
<p>As shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, a deeper analysis of the confusion matrix corresponding to the three predicted states was conducted. The results illustrated that the proposed method demonstrated remarkable predictive accuracy. For each ground-truth label across the three states, the majority of the predicted labels align with either the ground-truth labels or their neighboring labels (the cumulative probability for these exceeds 99%), which was reasonable given the inherent difficulty in distinguishing between neighboring state image features due to their significant similarity. It is difficult to capture the morphological changes at the critical stage of the tobacco leaf curing process. Even if experienced experts make judgments, they may still misjudge. The prediction of the browning degree (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8b</bold>
</xref>) demonstrates high accuracy across all degrees. However, further improvements are still required in the prediction of the middle degrees of yellowing (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8a</bold>
</xref>) and drying (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8c</bold>
</xref>). Generally, the accuracy of the proposed method complete recognition is 83%, the accuracy rate of adjacent stages is more than 99%, and the fault tolerance rate is within &#xb1;1 stage, which has little impact on the curing quality during the actual curing process.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>The confusion matrices for the prediction of the <bold>(A)</bold> Yellowing, <bold>(B)</bold> Browning and <bold>(C)</bold> Drying states of the tobacco leaves.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g008.tif">
<alt-text content-type="machine-generated">Three confusion matrices labeled A, B, and C, each representing different conditions: Yellowing, Browning, and Drying. Each matrix shows true versus predicted labels ranging from zero to six. The color intensity indicates accuracy, with values closer to one being darker blue, showing higher accuracy percentages along the diagonal.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>Visualization of the gradient.</title>
<p>In <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, GradCam (<xref ref-type="bibr" rid="B14">Selvaraju et&#xa0;al., 2017</xref>) was used to visualize the gradients for each state. The results showed that in this example, the model proposed in this study focused primarily on the brown parts of the image when predicting browning degree, on the petiole when predicting drying degree, and on larger areas of leaf content when predicting yellowing degree. This is similar to the reference positions and standards that people use to judge the three states of tobacco leaves during the curing process.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Visualization of gradients with respect to ground truth labels. <bold>(A)</bold> Original image, <bold>(B)</bold> Browning, <bold>(C)</bold> Drying and <bold>(D)</bold> Yellowing states.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-16-1604382-g009.tif">
<alt-text content-type="machine-generated">Panel of four images labeled A to D. (A) Shows yellowed leaves with some browning spots. (B) Depicts a thermal or spectral image with a red central area surrounded by blue. (C) Similar to B, with a larger red area, indicating more heat or activity. (D) Features an expanded red area, showing the most intense heat or activity depicted in the sequence.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusions</title>
<p>In this study, an large-scale dataset including 17,420 images of tobacco leaves from 10 main production areas in China was developed, with a specific emphasis on recognizing the degrees of yellowing, browning, and drying. This is a large-scale dataset specifically dedicated to the morphological recognition of the states of tobacco leaves within an actual bulk curing barn setting. A deep learning benchmark was then established for this dataset using various deep learning networks. To further enhance the predictive performance of the deep backbone network, an efficient deep learning method was proposed, including Fourier filter module and common color filter module. This method integrated the spectral characteristics of tobacco leaves images and filters out color noise, which effectively enhanced the accuracy of our model with prediction accuracy for the yellowing degree, browning degree, and drying degree were 83.0%, 90.5%, and 75.6% respectively. The high overall average accuracy with a value of 83.0% and the availability and feasibility in different production areas have demonstrated the superior performance of the proposed method in this study, which provides a solid foundation for future research in this area.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>QX: Conceptualization, Methodology, Project administration, Software, Writing &#x2013; original draft. YZ: Conceptualization, Funding acquisition, Resources, Writing &#x2013; review &amp; editing. AW: Data curation, Investigation, Writing &#x2013; original draft. GC: Data curation, Validation, Writing &#x2013; review &amp; editing. XC: Data curation, Validation, Writing &#x2013; review &amp; editing. SZ: Data curation, Validation, Writing &#x2013; review &amp; editing. JYL: Data curation, Validation, Writing &#x2013; review &amp; editing. BJ: Data curation, Validation, Writing &#x2013; review &amp; editing. DY: Data curation, Validation, Writing &#x2013; review &amp; editing. JH: Data curation, Validation, Writing &#x2013; review &amp; editing. ZC: Data curation, Validation, Writing &#x2013; original draft. HZ: Data curation, Validation, Writing &#x2013; review &amp; editing. JW: Data curation, Validation, Writing &#x2013; review &amp; editing. WG: Conceptualization, Data curation, Funding acquisition, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. JJL: Conceptualization, Data curation, Investigation, Validation, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by the foundation from the China National Tobacco Corp., No. 110202201051 (SJ-01) and No. 110202101084 (SJ-08), and the foundation from the Henan Provincial Tobacco Company No.2024410000240029. The authors declare that this study received funding from Henan Provincial Tobacco Company. The funder had the following involvement in the study: data collection.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Authors XC and DY were employed by the company Shanghai Tobacco Group Co. Ltd. Authors JYL and HZ were employed by the company Henan Provincial Tobacco Company. Author BJ was employed by the company China Tobacco Guangdong Industrial Co. Ltd. Author ZC was employed by the company Jilin Tobacco Industry Co. Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest. Author Contributions</p>
</sec>
<sec id="s9" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Condor&#xed;</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Albesa</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Altobelli</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Duran</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Sorrentino</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Image processing for monitoring of the cured tobacco process in a bulk-curing stove</article-title>. <source>Comput. Electron. Agric.</source> <volume>168</volume>, <fpage>105113</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2019.105113</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cover</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Hart</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>1967</year>). <article-title>Nearest neighbor pattern classification</article-title>. <source>IEEE Trans. Inf. Theory</source> <volume>13</volume>, <fpage>21</fpage>&#x2013;<lpage>27</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIT.1967.1053964</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dosovitskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Beyer</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Kolesnikov</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Weissenborn</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Unterthiner</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>An image is worth 16x16 words: Transformers for image recognition at scale</article-title>. <source>arXiv arXiv:2010.11929</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id> </citation>
</ref>
<ref id="B5">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <conf-name>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <conf-loc>Las Vegas, NV, USA</conf-loc>: <publisher-name>IEEE</publisher-name>. <fpage>770</fpage>&#x2013;<lpage>778</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hearst</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Dumais</surname> <given-names>S. T.</given-names>
</name>
<name>
<surname>Osuna</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Platt</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Scholkopf</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Support vector machines</article-title>. <source>IEEE Intelligent Syst. their Appl.</source> <volume>13</volume>, <fpage>18</fpage>&#x2013;<lpage>28</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/5254.708428</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Howard</surname> <given-names>A. G.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Kalenichenko</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Weyand</surname> <given-names>T.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Mobilenets: Efficient convolutional neural networks for mobile vision applications</article-title>. <source>arXiv arXiv:1704.04861</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1704.04861</pub-id> </citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Selection of optimum discriminant model in tobacco curing stage based on image processing</article-title>. <source>Acta Ta bacaria Sin.</source> <volume>28</volume>, <fpage>65</fpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). &#x201c;<article-title>Swin transformer v2: Scaling up capacity and resolution</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>. New <publisher-loc>Orleans, LA</publisher-loc>: <publisher-name>IEEE/CVF</publisher-name>. <fpage>12009</fpage>&#x2013;<lpage>12019</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2201.12086</pub-id> </citation>
</ref>
<ref id="B10">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). &#x201c;<article-title>Swin transformer: Hierarchical vision transformer using shifted windows</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF international conference on computer vision</conf-name>. <publisher-loc>Montreal, QC, Canada</publisher-loc>: <publisher-name>IEEE/CVF</publisher-name>. <fpage>10012</fpage>&#x2013;<lpage>10022</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Intelligent grading of tobacco leaves using an improved bilinear convolutional neural network</article-title>. <source>IEEE Access</source> <volume>11</volume>, <fpage>68153</fpage>&#x2013;<lpage>68170</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2023.3292340</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Gai</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Construction of benchmark data set for intelligent identific ation of tobacco pests,diseases, phytotoxicity and design of three dimensional attention model</article-title>. <source>Acta Tabacaria Sin.</source> <volume>27</volume>, <fpage>52</fpage>&#x2013;<lpage>60</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2106.07178</pub-id> </citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pei</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>State recognition and temperature rise time prediction of tobacco curing using multi-sensor data-fusion method based on feature impact factor</article-title>. <source>Expert Syst. Appl.</source> <volume>237</volume>, <fpage>121591</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2023.121591</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Selvaraju</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Cogswell</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Das</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Vedantam</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Parikh</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Batra</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Grad-cam: Visual explanations from deep networks via gradient-based localization</article-title>,&#x201d; in <conf-name>2017 IEEE International Conference on Computer Vision (ICCV)</conf-name>, <conf-loc>Venice, Italy</conf-loc>: <publisher-name>IEEE</publisher-name>. <fpage>618</fpage>&#x2013;<lpage>626</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Siddiqui</surname> <given-names>K. M.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Analysis of a malakisi barn used for tobacco curing in east and southern africa</article-title>. <source>Energy Conversion Manage.</source> <volume>42</volume>, <fpage>483</fpage>&#x2013;<lpage>490</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0196-8904(00)00066-2</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zisserman</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arXiv arXiv:1409.1556</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id> </citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname> <given-names>A. N.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). <article-title>Attention is all you need</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>30</volume>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1706.03762</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Intelligent tobacco flue-curing method based on leaf texture feature analysis</article-title>. <source>Optik</source> <volume>150</volume>, <fpage>117</fpage>&#x2013;<lpage>130</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ijleo.2017.09.088</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Research on state prediction method of tobacco curing process based on model fusion</article-title>. <source>J. Ambient Intell. Humanized Computing</source> <volume>13</volume>, <fpage>2951</fpage>&#x2013;<lpage>2961</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12652-021-03129-5</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S. X.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Intelligent control of bulk tobacco curing schedule using LS-SVM- and ANFIS-based multi-sensor data fusion approaches</article-title>. <source>Sensors</source> <volume>19</volume>, <fpage>1778</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s19081778</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S. X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Modeling of the bulk tobacco flue-curing process using a deep learning-based method</article-title>. <source>IEEE Access</source> <volume>9</volume>, <fpage>140424</fpage>&#x2013;<lpage>140436</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2021.3119544</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>S. X.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A novel intelligent control system for flue- curing barns based on real-time image features</article-title>. <source>Biosyst. Eng.</source> <volume>123</volume>, <fpage>77</fpage>&#x2013;<lpage>90</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.biosystemseng.2014.05.008</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>A novel integrated system for real-time monitoring of tobacco leaf images in the bulk curing barn</article-title>,&#x201d; in <source>2024 IEEE International Conference on Advanced Intelligent Mechatronics (AIM)</source> (<publisher-loc>Lyon, France</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1573</fpage>&#x2013;<lpage>1578</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Intelligent tobacco curing control based on color recognition</article-title>. <source>Res. J. Appl. Sciences Eng. Technol.</source> <volume>5</volume>, <fpage>2509</fpage>&#x2013;<lpage>2513</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.19026/rjaset.5.4688</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>X.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>In-field tobacco leaf maturity detection with an enhanced mobilenetv1: Incorporating a feature pyramid network and attention mechanism</article-title>. <source>Sensors</source> <volume>23</volume>, <fpage>5964</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s23135964</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Recognition of tobacco leaf curing stage based on deep learning</article-title>,&#x201d; in <conf-name>2024 10th International Conference on Electrical Engineering, Control and Robotics (EECR)</conf-name>, <conf-loc>Guangzhou, China</conf-loc>: <publisher-name>EEE</publisher-name>. <fpage>305</fpage>&#x2013;<lpage>309</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>