<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2024.1500819</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Detection of soluble solids content in tomatoes using full transmission Vis-NIR spectroscopy and combinatorial algorithms</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Cai</surname>
<given-names>Letian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhang</surname>
<given-names>Yizhi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cai</surname>
<given-names>Zhonglei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shi</surname>
<given-names>Ruiyao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Sheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1804493"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Jiangbo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1555386"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Intelligent Equipment Research Center, Beijing Academy of Agriculture and Forestry Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>National Agricultural Intelligent Equipment Engineering Technology Research Center</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Justyna Cybulska, Polish Academy of Sciences, Poland</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Xudong Sun, East China Jiaotong University, China</p>
<p>Xiaping Fu, Zhejiang Sci-Tech University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Jiangbo Li, <email xlink:href="mailto:lijb@nercita.org.cn">lijb@nercita.org.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>11</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>15</volume>
<elocation-id>1500819</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>09</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>24</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Cai, Zhang, Cai, Shi, Li and Li</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Cai, Zhang, Cai, Shi, Li and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Soluble solids content (SSC) is an important indicator for evaluating tomato flavor, and general physical and chemical methods are time-consuming and destructive.</p>
</sec>
<sec>
<title>Methods</title>
<p>This study utilized full transmittance visible and near infrared (Vis-NIR) spectroscopy for multi-posed data acquisition of tomatoes in different orientations. The role of two directions (Z1 and Z2) and four preprocessing techniques, as well as three wavelength selection methods in the exploitation of SSC regression models was investigated.</p>
</sec>
<sec>
<title>Results</title>
<p>After using the Outlier elimination method, the spectra acquired in the Z2 direction and the raw spectral data processed by preprocessing methods gave the best result by the PLSR model (<italic>R<sub>p</sub>
</italic> = 0.877, <italic>RMSEP</italic> = 0.417 %). Compared to the model built using the full 2048 spectral wavelengths, the prediction accuracy using 20 wavelengths obtained by a combination wavelength selection: backward variable selection - partial least squares and simulated annealing (BVS-PLS and SA) was further improved (<italic>R<sub>p</sub>
</italic> = 0.912, <italic>RMSEP</italic> = 0.354 %).</p>
</sec>
<sec>
<title>Discussion</title>
<p>The findings of this research demonstrate the efficacy of full-transmission visible-near infrared (Vis-NIR) spectroscopy in forecasting SSC of tomatoes, and most importantly, the combination of the packing method in wavelength selection with an intelligent optimization algorithm provides a viable idea for accurately and rapidly assessing the SSC of tomatoes.</p>
</sec>
</abstract>
<kwd-group>
<kwd>tomato</kwd>
<kwd>online detection</kwd>
<kwd>feature selection</kwd>
<kwd>internal quality assessment</kwd>
<kwd>soluble solids content</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="4"/>
<equation-count count="2"/>
<ref-count count="31"/>
<page-count count="12"/>
<word-count count="5566"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Crop and Product Physiology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Tomatoes, known as the fruit of vegetables, are widely grown around the world, and their consumption helps to consume fiber, antioxidants and a variety of minerals, which can reduce the likelihood of developing cancer and chronic illnesses, so tomatoes are popular with the public (<xref ref-type="bibr" rid="B20">Ozturkoglu-Budak and Aksahin, 2016</xref>). With the increase of tomato demand, the quality of tomato has been paid more and more attention. High quality fresh tomatoes require high nutritional value and good taste. Soluble solid content (SSC) refers to the percentage of soluble substances such as soluble sugar and organic acid in tomatoes, which is a very important indicator to measure the internal quality of tomatoes and is closely associated with consumers&#x2019; perceptions of the intrinsic quality characteristics of the fruit (<xref ref-type="bibr" rid="B31">Zheng et&#xa0;al., 2024</xref>). The conventional approach for measuring the SSC of tomatoes typically employs the refractometer technique, which necessitates the extraction of juice from the fruit followed by titration. This procedure is not only time-intensive but also destructive, rendering it impractical for large-scale fruit analysis (<xref ref-type="bibr" rid="B25">Tan et&#xa0;al., 2022</xref>). Consequently, the development of a non-destructive and efficient measurement technique for the quality assessment of tomatoes is of considerable importance.</p>
<p>Spectral analysis technology is usually used to study the relationship between light and matter, and obtain spectral information through the response of matter to light, so as to reflect the physical or chemical information of the target region. Visible near-infrared (Vis-NIR) spectroscopy analysis is one of the mainstream methods for non-destructive examination of fruit internal quality (<xref ref-type="bibr" rid="B18">Mei and Li, 2023</xref>). <xref ref-type="bibr" rid="B27">Zhang et&#xa0;al. (2021)</xref> achieved accurate prediction of tomato SSC by setting different Vis-NIR spectral ranges. <xref ref-type="bibr" rid="B1">Brito et&#xa0;al. (2021)</xref> collected the NIR spectral data of tomato and established a partial least squares (PLS) regression model for the detection of SSC of tomato by using orthogonal signal correction, and the standard deviations of the calibration and validation sets were obtained to be 0.52% and 0.56%, respectively. <xref ref-type="bibr" rid="B9">Huang et&#xa0;al. (2018)</xref> used a self-constructed system for spatially resolved spectroscopy to detect tomato quality, and the results proved that the system had a significant advantage over the traditional single-point Vis/NIRS instrument in tomato SSC assessment, with <italic>Rp</italic> and <italic>RMSEP</italic> of 0.801 and 0.38%, respectively. For the characteristics of the heterogeneous structure of tomato, it is necessary to obtain as much information as possible about the interior of the tomato, however, typical Vis-NIR spectroscopy detection technique is flawed because single-point measurements are capable of providing only a restricted amount of spatial information regarding the sample. <xref ref-type="bibr" rid="B26">Yang et&#xa0;al. (2022)</xref> optimized the optical path, light intensity and other detection settings of the Vis-NIR diffuse transmittance system and developed a compensation model based on the physiological traits of tomato to obtain a favorable accuracy of SSC detection in tomato (Rp = 0.91, PMSEP = 0.17%). In contrast to the reflection mode, the transmission mode is capable of providing a greater amount of information regarding the internal structure and material of the tomato fruit; in addition, the implementation of full transmission and continuous data acquisition methodologies can address the constraints associated with the conventional single-point Vis-NIR measurement technique, thereby facilitating a thorough characterization of the entire tomato&#x2019;s properties.</p>
<p>Owing to advancements in contemporary analytical methodologies, Vis-NIR spectroscopy makes it easy to measure information about objects characterized by a large number of spectral bands in a short period of time. Pluralistic metrological techniques were used to extract the most useful information from redundant data, and enhanced quantitative calibration models were developed through the systematic evaluation of characteristic wavelengths or wavelength intervals through variable selection methods. Numerous studies in the fields of statistics and data analysis elucidate various methodologies for the selection of variables which can be broadly categorized into three groups: filtering, wrapping and embedding. Filtering method selects variables and evaluates them independently by introducing thresholds (e.g., load weights or regression coefficients). <xref ref-type="bibr" rid="B16">Liu et&#xa0;al. (2008)</xref> used NIR spectroscopy to differentiate fruit vinegar varieties based on a filtering method. <xref ref-type="bibr" rid="B10">Inoue et&#xa0;al. (2012)</xref> used a correlation coefficient filtering method to accurately assess the nitrogen content of rice canopies. The wrapping method takes into account the correlation between the variables and selects them by evaluating the impact of the combination of variables on the model performance. <xref ref-type="bibr" rid="B2">Cai et&#xa0;al. (2008)</xref> showed that the uninformative variable elimination (UVE) - PLS wrapping method could more accurately predict nicotine content in tobacco samples. The embedding method involves the selection of variables while building the model, and the interplay between variable selection and sample categorization leads to a reduction in the time required for analysis. <xref ref-type="bibr" rid="B19">Ning et&#xa0;al. (2022)</xref> used the least absolute shrinkage and selection operator (LASSO) embedding method to select variables from preprocessed NIR spectra, and realized the quantitative detection of mycotoxins in wheat kernels by develop a SVM model. The above analysis indicated that the amount of data can be greatly compressed by using different variable selection and its combination algorithms, thus the operational efficiency of the model was enhanced; concurrently, the predictive accuracy and stability of the model were further augmented through the removal of nonlinear or extraneous variables.</p>
<p>In conclusion, the primary objective of this research was to develop an optimal model for the detection of SSC in tomatoes utilizing full transmission Vis-NIR spectroscopy. The specific aims of the study were as follows: (1) To evaluate the effect of different tomato placement orientations on spectral prediction accuracy; (2) To investigate the effect of different spectral pretreatment methods on tomato transmission spectrum; (3) To screen the spectral wavelengths using different feature selection algorithms, and determine the most effective predictive model by evaluating both its accuracy and the time required for modeling.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Sample preparation</title>
<p>Ninety samples of fresh tomatoes (&#x2018;Provence&#x2019; variety) were purchased from a vegetable supermarket in Beijing, China. &#x2018;Provence&#x2019; tomatoes have thin skins, juicy flesh, and a rich reddish color when ripe, and the used samples were free of any surface damage. To mitigate temperature variations that may lead to inaccuracies in measurements, the tomatoes were maintained at a temperature of 20&#xb0;C and a relative humidity of 60% for a duration of 24 hours prior to the collection of spectral and SSC data. During the development of the model, the ratio of sample numbers between the calibration set and the prediction set was established at 3:1.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Online full transmission spectrum acquisition system</title>
<p>The spectral data of the tomato samples were collected by using a full transmission Vis-NIR online detection system shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, and the main units include: a highly sensitive spectrometer (wavelength range: 560-1072 nm, wavelength interval: 0.25 nm, integration time: 5 ms), a speed-adjustable moving platform, a dark box, relative position sensors, an illumination unit consisting of a 150 W halogen lamp with focusing and attenuating device and a computer for the control system. The illumination unit and the spectrometer are placed 150 mm apart on each side of the conveyor belt in the dark box.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Full transmission Vis-NIR online tomato detection system.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1500819-g001.tif"/>
</fig>
<p>In order to assess the effect of the complex cavity inside the tomato on the accuracy of the on-line detection of SSC, each sample passed through the measuring device on a conveyor belt in two different orientations, Z1 orientation: the stem-calyx axis of the test samples were perpendicular to the conveyor belt, and the stems were facing upward; Z2 orientation: the stem-calyx axis of the test samples were parallel to the conveyor belt, and the stems were facing toward the spectrometer, and the orientation of the test samples were shown schematically in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2A</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>
<bold>(A)</bold> Tomato detection orientations. <bold>(B)</bold> Multi-point raw spectral curves. <bold>(C)</bold> Spectral curve after averaging of multi-point raw data.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1500819-g002.tif"/>
</fig>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>SSC measurement</title>
<p>SSC refers to the percentage of solvable content in the tomato juice, which is mainly composed of nutrients such as soluble sugar (C&#x2086;H&#x2081;&#x2082;O&#x2086;) and organic acids (-COOH), and is an important indicator of tomato quality and fruit processing characteristics. After online spectrum acquisition, the traditional destructive method was used to measure the SSC of tomatoes immediately. The measuring instrument was digital Abbe refractometer (model PAL-1, Atago Co., Tokyo, Japan). The intact tomatoes were first chopped and pressed in a wall breaker to make tomato juice, then the tomato juice was filtered through gauze and the juice was dripped into a beaker. After sufficient shaking, about 0.3 ml of juice was dripped onto a digital refractometer (zero-corrected) and the SSC value was recorded manually.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Data preprocessing</title>
<p>At the time when the tomato samples just reached the detection position and left the detection position, the light signal from the spectrometer passed through only a small portion of the sample&#x2019;s pericarp tissue. Due to the high optical intensity, these portions of the spectrum should be removed prior to modeling. The extent of removal in this study was about halfway between the two sides of the sample pericarp, which is close to the ideal range. The concluding spectral data obtained encompassed the majority of the pulp and cavity regions within each tomato sample, representing the primary internal quality areas of the entire tomato.</p>
<p>To further improve the spectral data quality, the pre-processing algorithms were generally used to reduce the noise and interference of the instrument and background before building the prediction model. Gaussian filter (GF) is a linear smoothing filter based on the Gaussian function (<xref ref-type="bibr" rid="B29">Zhang et&#xa0;al., 2024</xref>), The processing process is as follows: Firstly, the filter parameters are determined, then the GF is constructed, the weight of each data point in the filter is calculated and normalized according to the window size and standard deviation, and finally the data is convolved. Due to the influence of instrument accuracy and background noise, the spectral signals obtained by the spectrometer contain both useful information and random errors superimposed on them at the same time, and the use of smoothing algorithms reduces the noise and improves the signal-to-noise ratio (<xref ref-type="bibr" rid="B3">Chen et&#xa0;al., 2014</xref>). Standard normal variate (SNV) is mainly used to eliminate the effects of surface scattering as well as optical range changes (<xref ref-type="bibr" rid="B4">Dhanoa et&#xa0;al., 2023</xref>). The processing process is as follows: Firstly, for a given set of spectral data, the mean and standard deviation of all wavelength points in the spectrum are calculated; the data value for each wavelength point is then subtracted from the mean of the sample spectrum and divided by the standard deviation. Multiplicative scatter correction (MSC) is primarily employed to mitigate the scattering effects arising from heterogeneous particle distribution and variations in particle size (<xref ref-type="bibr" rid="B14">Li et&#xa0;al., 2018</xref>). The processing process is as follows: Firstly, a spectrum considered representative is selected as the reference spectrum. Then, for each spectrum to be processed, the linear relationship between it and the reference spectrum is calculated. Finally, the data of each wavelength point of the spectrum are corrected by using the obtained linear equation parameters. In this study, GF, SNV and SMC preprocessing algorithms were used to refine the tomato full transmission spectrum data.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Prediction model and evaluation indicators</title>
<p>Partial least squares regression (PLSR) is a multivariate factorial regression technique frequently employed in the field of spectral analysis. PLSR constructs predictive models by finding the optimal linear combinations of independent and dependent variables and extracting the latent variables (<italic>LVs</italic>) that maximize the correlation between input and output variables (<xref ref-type="bibr" rid="B5">Diniz et&#xa0;al., 2015</xref>). In addition, PLSR demonstrates computational efficiency, particularly in scenarios where the sample size is limited while the number of variables is extensive (<xref ref-type="bibr" rid="B12">Li et&#xa0;al., 2023</xref>). In this research, a PLSR model was developed to elucidate the quantitative relationship between the spectral matrix of tomatoes (X) and the matrix of SSC values (Y). The root mean square error of cross-validation (<italic>RMSECV</italic>) was employed to ascertain the optimal number of <italic>LVs</italic>.</p>
<p>Calibration correlation coefficient (<inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), root mean square error of calibration (<italic>RMSEC</italic>), prediction correlation coefficient (<inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), and root mean square error of prediction (<italic>RMSEP</italic>) were used to assess model performance. See <xref ref-type="disp-formula" rid="eq1">Equation 1</xref> for specific calculations. Typically, models with higher correlation coefficients (<inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) and lower root-mean-square errors (<italic>RMSEC</italic>, <italic>RMSECV</italic>, and <italic>RMSEP</italic>) can be considered to meet expectations. Matlab 2023b (Mathworks, Natick, MA) performed the development of all model programs.</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">^</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msqrt>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">^</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">^</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="true">&#xaf;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> denote the measured, predicted and mean values of the <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> tomato sample in the calibration or prediction set, respectively, and <inline-formula>
<mml:math display="inline" id="im11">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> represents the total quantity of tomato samples included in either the calibration or prediction dataset.</p>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Wavelength selection methods</title>
<p>The full spectrum contains 2048 wavelengths with a large number of uncorrelated and co-linear variables. Moreover, this adds complexity to the model, a large number of wavelengths may introduce interference, which can make the model run slower and less accurate (<xref ref-type="bibr" rid="B17">Luo et&#xa0;al., 2022</xref>). Therefore, it is important to find a variable selection algorithm to simplify the spectral data without reducing the accuracy of the model (<xref ref-type="bibr" rid="B13">Li et&#xa0;al., 2024</xref>).</p>
<sec id="s2_6_1">
<label>2.6.1</label>
<title>Backward variable selection PLS</title>
<p>BVS-PLS is a packaging technique that utilizes the PLS regression algorithm and operates iteratively through a process of backward selection (<xref ref-type="bibr" rid="B7">Fern&#xe1;ndez et&#xa0;al., 2009</xref>). In this study, <italic>RMSECV</italic> was used as a criterion for variable optimization, which was calculated as follows:</p>
<p>Step-1: The <inline-formula>
<mml:math display="inline" id="im12">
<mml:mi>w</mml:mi>
</mml:math>
</inline-formula> full-spectrum wavelengths were divided into <inline-formula>
<mml:math display="inline" id="im13">
<mml:mi>m</mml:mi>
</mml:math>
</inline-formula> groups, each containing <inline-formula>
<mml:math display="inline" id="im14">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> number of wavelengths, <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
<p>Step-2: The PLS regression model was fitted to a grouped dataset containing all <inline-formula>
<mml:math display="inline" id="im16">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> wavelengths and <italic>RMSECV</italic> was calculated as the initial value for the iteration.</p>
<p>Step-3: Cycled through the different groups. Deleted one wavelength group at a time and fit the PLS model to the remaining <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> wavelength groups, with the wavelength group with the largest decrease in <italic>RMSECV</italic> being discarded.</p>
<p>Step-4: Step-3 was repeated, iteratively discarding wavelength sets until a point was reached where the <italic>RMSECV</italic> was no longer decreasing. At this point all wavelengths with poor correlation with the tomato SSC to be predicted have been eliminated.</p>
<p>During the algorithmic loop, the data was tested using leave-one-out cross-validation.</p>
</sec>
<sec id="s2_6_2">
<label>2.6.2</label>
<title>Simulated annealing algorithm</title>
<p>Simulated annealing algorithm is a widely used heuristic stochastic intelligent optimization algorithm (<xref ref-type="bibr" rid="B6">Ekinci et&#xa0;al., 2023</xref>), SA distinguishes itself from other optimization algorithms in that it can receive solutions that are worse than the previous result (<xref ref-type="bibr" rid="B24">Suman and Kumar, 2006</xref>; <xref ref-type="bibr" rid="B15">Lin et&#xa0;al., 2018</xref>), and thus is able to obtain a more varied solution space that is not prone to falling into local optima (<xref ref-type="bibr" rid="B22">Shi et&#xa0;al., 2024</xref>). The specific calculation steps are as follows:</p>
<p>Step-1: Initialized the annealing table, which consisted of the initial temperature <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the cooling parameter <inline-formula>
<mml:math display="inline" id="im19">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula> in the temperature update function, the maximum number of iterations <inline-formula>
<mml:math display="inline" id="im20">
<mml:mi>L</mml:mi>
</mml:math>
</inline-formula>, and the termination temperature <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Step-2: Randomly generated a <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as the current solution <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Step-3: Generated a neighborhood solution <inline-formula>
<mml:math display="inline" id="im24">
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula>. The selection rule of the neighborhood solution was: if the objective function was continuous, generated a random vector <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>; if the objective function was discrete, generated a random offset <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and obtained a neighborhood solution by <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>z</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Step-4: Calculated and compared the fitness functions <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, if <inline-formula>
<mml:math display="inline" id="im29">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&lt;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, received <inline-formula>
<mml:math display="inline" id="im30">
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula>; if <inline-formula>
<mml:math display="inline" id="im31">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&gt;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, extracted a random number from a uniform distribution in [0,1] that was less than the probability value and accepted the change <inline-formula>
<mml:math display="inline" id="im32">
<mml:msup>
<mml:mi>h</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula>.</p>
<p>Step-5: Determined whether the algorithm reaches the maximum number of iterations <inline-formula>
<mml:math display="inline" id="im33">
<mml:mi>L</mml:mi>
</mml:math>
</inline-formula>, if it meets, then go to step-6, if not, then return to step-3.</p>
<p>Step-6: Ascertain whether the termination criterion has been met; if it has been met, present the optimal solution; if it was not satisfied, update the temperature using the temperature update function <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>T</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and return to step-3.</p>
</sec>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and discussion</title>
<sec id="s3_1">
<label>3.1</label>
<title>Influence of spectral acquisition orientations on the prediction of tomato SSC</title>
<p>Tomato is usually not a completely uniform structure. There may be differences in the internal organization of different parts of the tomato, such as the cell structure, water distribution, sugar content, etc. near the stem and in the middle of the tomato. By studying different spectral acquisition locations, the characteristics of different regions inside the tomato can be more comprehensively understood, so as to find the most suitable spectral acquisition strategy for specific tomato fruit varieties, and improve the universality and accuracy of the prediction model. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref> showed the original spectral data collected by multi-point full transmission measurement way. It can be seen that the intensity of each spectral curve was relatively large. At the same time, due to the different acquisition locations and the variability of physical and chemical properties of tomatoes, there were obvious intensity differences between different spectral curves. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2C</bold>
</xref> showed the spectral curve after averaging the original data from multiple points. Generally speaking, Z1 orientation: the stem-calyx axis of the test samples were perpendicular to the conveyor belt, and Z2 orientation: the stem-calyx axis of the test samples were parallel to the conveyor belt, the trend and characteristics of the spectral curves obtained by the two acquisition methods were basically similar, but the intensity of the optical signal collected in the orientation of Z2 was higher than that of Z1, this phenomenon can be attributed to the impact of the internal cavity architecture of the tomato sample on the trajectory of light propagation. Due to the shorter optical path distance in the Z2 direction, which was less affected by the structure of the tomato cavity, the transmission spectral signals obtained were stronger.</p>
<p>For the raw transmission spectral data collected in different orientations, the corresponding PLSR models were established to evaluate their prediction performance. As can be seen from <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, for Z1 orientation, <inline-formula>
<mml:math display="inline" id="im37">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <italic>RMSEC</italic> of model were 0.951 and 0.258%, respectively, <inline-formula>
<mml:math display="inline" id="im38">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <italic>RMSEP</italic> were 0.637 and 0.713%, respectively; and for Z2 orientation, <inline-formula>
<mml:math display="inline" id="im39">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <italic>RMSEC</italic> of model were 0.917 and 0.338%, respectively, <inline-formula>
<mml:math display="inline" id="im40">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <italic>RMSEP</italic> were 0.757 and 0.583%, respectively. Compared with Z1, the model built in the Z2 orientation improved the accuracy of the prediction set while avoiding potential overfitting of the correction set. For the two detection orientations, the performance of the model was also improved with the increase of spectral intensity. Based on the above analysis, the raw spectral data obtained in the Z2 orientation were used for the subsequent modeling steps.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>SSC of tomatoes was predicted by full spectrum PLS model in different acquisition orientations.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Acquisition orientation</th>
<th valign="middle" align="left">
<italic>LVs</italic>
</th>
<th valign="middle" align="left">
<inline-formula>
<mml:math display="inline" id="im35">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th valign="middle" align="left">
<italic>RMSEC (%)</italic>
</th>
<th valign="middle" align="left">
<inline-formula>
<mml:math display="inline" id="im36">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th valign="middle" align="left">
<italic>RMSEP (%)</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Z1</td>
<td valign="middle" align="left">11</td>
<td valign="middle" align="left">0.951</td>
<td valign="middle" align="left">0.258</td>
<td valign="middle" align="left">0.637</td>
<td valign="middle" align="left">0.713</td>
</tr>
<tr>
<td valign="middle" align="left">Z2</td>
<td valign="middle" align="left">10</td>
<td valign="middle" align="left">0.917</td>
<td valign="middle" align="left">0.338</td>
<td valign="middle" align="left">0.757</td>
<td valign="middle" align="left">0.583</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Z1: the stem-calyx axis of the test samples was perpendicular to the conveyor belt, Z2: the stem-calyx axis of the test samples was parallel to the conveyor belt.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Spectrum preprocessing and abnormal sample removal</title>
<p>The spectrum obtained by the instrument encompasses not only the chemical characteristics of the sample but also extraneous information and noise, including electrical interference, background signals from the sample, and stray light. In this study, four preprocessing methods, including MSC, SNV, GF and GF and MSC, were used to process the original transmission spectrum of the tomato samples, and the spectral curves after preprocessing were shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. It can be seen that the MSC and SNV algorithms reduced the influence of the heterogeneous particle distribution in tomato. Compared with the original spectra in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2B</bold>
</xref>, the spectral curves after GF treatment reduced the undesirable noise and presented smooth curves in the figure. The spectral curves after GF and MSC treatment also showed the same spectral trend as in the above analysis.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Spectral preprocessing of all samples.<bold>(A)</bold> MSC, <bold>(B)</bold> SNV, <bold>(C)</bold> GF, and <bold>(D)</bold> GF and MSC.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1500819-g003.tif"/>
</fig>
<p>The results in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> showed that the <inline-formula>
<mml:math display="inline" id="im43">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>was improved after the pre-processing of MSC and SNV, and the <italic>RMSEC</italic> showed a significant decrease, which may be the result of suppressing the surface scattering of the samples. However, the improvement of the model performance in the prediction set was limited. After the GF treatment, although the performance of the model on the calibration set appeared to be degraded, the accuracy of the model on the prediction set was significantly improved. As a comparison, after GF and MSC combination preprocessing, model obtained the better performance (<inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>= 0.927, <italic>RMSEC</italic> = 0.316%, <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> = 0.852, <italic>RMSEP</italic> = 0.456%). Obviously, the GF and MSC preprocessing process amplified the spectral properties and resulted in clearer and more consistent spectra, which in turn improved the stability of the data.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Prediction results of tomato SSC by PLSR model with preprocessed full wavelengths.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Number of samples</th>
<th valign="middle" align="left">Methods</th>
<th valign="middle" align="left">
<italic>LVs</italic>
</th>
<th valign="middle" align="left">
<inline-formula>
<mml:math display="inline" id="im41">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th valign="middle" align="left">
<italic>RMSEC (%)</italic>
</th>
<th valign="middle" align="left">
<inline-formula>
<mml:math display="inline" id="im42">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th valign="middle" align="left">
<italic>RMSEP (%)</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="5" align="left">96 samples</td>
<td valign="middle" align="left">RAW</td>
<td valign="middle" align="left">10</td>
<td valign="middle" align="left">0.917</td>
<td valign="middle" align="left">0.338</td>
<td valign="middle" align="left">0.757</td>
<td valign="middle" align="left">0.583</td>
</tr>
<tr>
<td valign="middle" align="left">MSC</td>
<td valign="middle" align="left">11</td>
<td valign="middle" align="left">0.976</td>
<td valign="middle" align="left">0.184</td>
<td valign="middle" align="left">0.786</td>
<td valign="middle" align="left">0.547</td>
</tr>
<tr>
<td valign="middle" align="left">SNV</td>
<td valign="middle" align="left">12</td>
<td valign="middle" align="left">0.974</td>
<td valign="middle" align="left">0.189</td>
<td valign="middle" align="left">0.784</td>
<td valign="middle" align="left">0.550</td>
</tr>
<tr>
<td valign="middle" align="left">GF</td>
<td valign="middle" align="left">12</td>
<td valign="middle" align="left">0.845</td>
<td valign="middle" align="left">0.461</td>
<td valign="middle" align="left">0.831</td>
<td valign="middle" align="left">0.487</td>
</tr>
<tr>
<td valign="middle" align="left">GF+MSC</td>
<td valign="middle" align="left">12</td>
<td valign="middle" align="left">0.927</td>
<td valign="middle" align="left">0.316</td>
<td valign="middle" align="left">0.852</td>
<td valign="middle" align="left">0.456</td>
</tr>
<tr>
<td valign="middle" align="left">91 samples</td>
<td valign="middle" align="left">GF+MSC</td>
<td valign="middle" align="left">13</td>
<td valign="middle" align="left">0.939</td>
<td valign="middle" align="left">0.291</td>
<td valign="middle" align="left">0.877</td>
<td valign="middle" align="left">0.417</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In addition, the presence of outliers within the dataset can significantly impact the characteristics of the normal samples, which leads to the inaccuracy of the established model. Before further modeling, the relevant algorithm should be used to remove the outliers from the original sample (<xref ref-type="bibr" rid="B11">Lepot et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B23">Song et&#xa0;al., 2022</xref>). A Monte Carlo outlier detection approach was employed to identify potential outliers within the samples. This method leverages the mean and standard deviation (STD) of the prediction error, facilitating the detection and exclusion of outliers from both spectral data and SSC (<xref ref-type="bibr" rid="B28">Zhang et&#xa0;al., 2020</xref>). As shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, samples 24, 44, 46, 47, 48 were identified as potential outliers. To validate the accuracy of the algorithm, a PLS model was developed for SSC prediction of tomato samples based on the dataset before and after removal of each potential outlier. The results were shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. It can be seen that the prediction accuracies of the correction set and prediction set had been further improved (<inline-formula>
<mml:math display="inline" id="im46">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>= 0.939, <italic>RMSEC</italic> = 0.291%, <inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> = 0.877, <italic>RMSEP</italic> = 0.417%) after outlier removal. This suggests that the elimination of sample outliers diminished the variability within the data set, thereby enhancing the outcomes of the modeling and quantitative analysis.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Utilization of the Monte Carlo method for the identification of outliers.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1500819-g004.tif"/>
</fig>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>SSC prediction of tomatoes based on effective wavelengths</title>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Effective wavelength selection by BVS-PLS</title>
<p>Each spectral curve collected from tomato samples contained 2048 wavelengths, and adjacent wavelengths had similar spectral characteristics. The selection of optimal wavelengths from comprehensive spectral variables can reduce model complexity and enhance the accuracy of detecting SSC in tomatoes. Too many wavelengths will not only lead to multicollinearity, but also increase the running time of the model. Therefore, it is essential to examine the influence of wavelengths on the model.</p>
<p>Too many wavelengths not only have collinearity problems, but also make the data very complicated and increase the modeling time. Since the spectra of near wavelengths reflect similar physical and chemical properties of substances, the study attempted to improve the performance of the model by dividing the full spectral bands into different groups. To examine the impact of the quantity of spectral variables on the precision of regression analysis, the spectrum containing 2048 wavelength variables was divided into several segments <inline-formula>
<mml:math display="inline" id="im48">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> with the growth degree of 32, 16 and 8, respectively, and represented by the sum value of each segment, the spectrum with the number of wavelength variables <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> of 64, 128 and 256 was obtained. The recalculated spectrum was fed into the BVS-PLS algorithm, and the variables were iteratively deleted according to the <italic>RMSECV</italic> value. As can be seen from <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, after several iterations, the minimum <italic>RMSECV</italic> values at 64, 128 and 256 spectral wavelengths were 0.469%, 0.298% and 0.415%, respectively. In general, the value of <italic>RMSECV</italic> decreased first and then increased, which can be attributed to the fact that the more wavelength variables, the more information of the measured object was contained in the spectrum, which was conducive to reducing the regression error; however, more wavelength variables also brought more noise, resulting in reduced accuracy. Therefore, the combined 128 wavelengths were finally used in this study to establish a tomato SSC prediction model. The iterative process of the model was shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. For each iteration, one band was deleted and 59 wavelengths were finally selected. The selected bands were used for regression analysis, and the correlation coefficients on the calibration set and prediction set were 0.955 and 0.906, respectively, and the root-mean-square errors were 0.249 and 0.369, respectively. In comparison to the full-wavelength PLS model, the predictive performance of the model enhanced following variable selection through the BVS-PLS algorithm. This suggests that the process of wavelength selection contributes positively to the optimization of the model.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>The minimum <italic>RMSECV</italic> values of BVS-PLS based on the number of different variables.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Group</th>
<th valign="middle" align="left">Number of groups</th>
<th valign="middle" align="left">The number of variables</th>
<th valign="middle" align="left">
<italic>LVs</italic>
</th>
<th valign="middle" align="left">Minimum <italic>RMSECV (%)</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Group I</td>
<td valign="middle" align="left">64</td>
<td valign="middle" align="left">32</td>
<td valign="middle" align="left">15</td>
<td valign="middle" align="left">0.469</td>
</tr>
<tr>
<td valign="middle" align="left">Group II</td>
<td valign="middle" align="left">128</td>
<td valign="middle" align="left">16</td>
<td valign="middle" align="left">17</td>
<td valign="middle" align="left">0.298</td>
</tr>
<tr>
<td valign="middle" align="left">Group III</td>
<td valign="middle" align="left">256</td>
<td valign="middle" align="left">8</td>
<td valign="middle" align="left">20</td>
<td valign="middle" align="left">0.415</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>The trend of <italic>RMSECV</italic> values with increasing number of BVS-PLS iterations.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1500819-g005.tif"/>
</fig>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Effective wavelength selection by SA algorithm</title>
<p>SA algorithm is a global optimization algorithm, other evolutionary methods such as genetic algorithm and particle swarm optimization may easily fall into the local optimal solution during the search process, especially in the complex high-dimensional spectral wavelength space. The SA algorithm accepts poor solutions with a certain probability, which gives it a greater chance to jump out of the local optimal and explore a wider solution space in the search process, so it is more likely to find the global optimal feature combination. In addition, the SA algorithm has better convergence. In the process of decreasing temperature, the algorithm gradually stabilizes and tends to the optimal solution. In contrast, some evolutionary methods may be deficient in convergence speed and stability, especially when dealing with large-scale spectral wavelength data. In this study, the initial temperature <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was set to the initial <italic>RMSE</italic> value and the cooling parameter <inline-formula>
<mml:math display="inline" id="im51">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula> was set to 0.5% of the minimum <italic>RMSE</italic>. Note that the algorithm design avoided the use of a fixed value for <inline-formula>
<mml:math display="inline" id="im52">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula>. This has the advantage that the cooling parameter decreases as the <italic>RMSE</italic> decreases, thus providing more flexibility in finding a globally optimal solution. The maximum number of iterations <inline-formula>
<mml:math display="inline" id="im53">
<mml:mi>L</mml:mi>
</mml:math>
</inline-formula> was set to 500, and the termination temperature <inline-formula>
<mml:math display="inline" id="im54">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> was set to a value infinitely close to zero.</p>
<p>The number of wavelengths selected by SA algorithm was set to 20, 30, 40 and 50 respectively, and the influence of different wavelengths on the SSC prediction accuracy of tomato was tested. As can be seen from <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, at 20 wavelengths, <inline-formula>
<mml:math display="inline" id="im55">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>= 0.921, <italic>RMSEC</italic> = 0.330%, <inline-formula>
<mml:math display="inline" id="im56">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>= 0.889, <italic>RMSEP</italic> = 0.398%; at 50 wavelengths, <inline-formula>
<mml:math display="inline" id="im57">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>= 0.961, <italic>RMSEC</italic> = 0.232%, <inline-formula>
<mml:math display="inline" id="im58">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>= 0.932, <italic>RMSEP</italic> = 0.306%. In general, as the quantity of wavelengths increased, the model&#x2019;s accuracy on both the calibration and prediction datasets consistently enhanced, which was attributed to the fact that the more the number of spectral wavelengths, the richer the material information carried, this finding aligns with the analysis presented in section 3.3.1. However, too many wavelengths can make the prediction time of the model longer, which is not conducive to the need for online detection. Therefore, it is necessary to reduce the number of wavelengths as much as possible under the premise of ensuring the prediction accuracy of the model.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Prediction results of SSC in tomatoes by PLSR model with different wavelengths selected by SA algorithm.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Number of wavelengths</th>
<th valign="middle" align="left">
<italic>LVs</italic>
</th>
<th valign="middle" align="left">
<italic>Rc</italic>
</th>
<th valign="middle" align="left">
<italic>RMSEC (%)</italic>
</th>
<th valign="middle" align="left">
<italic>Rp</italic>
</th>
<th valign="middle" align="left">
<italic>RMSEP (%)</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">20</td>
<td valign="top" align="left">11</td>
<td valign="top" align="left">0.921</td>
<td valign="top" align="left">0.330</td>
<td valign="top" align="left">0.889</td>
<td valign="top" align="left">0.398</td>
</tr>
<tr>
<td valign="top" align="left">30</td>
<td valign="top" align="left">16</td>
<td valign="top" align="left">0.943</td>
<td valign="top" align="left">0.278</td>
<td valign="top" align="left">0.904</td>
<td valign="top" align="left">0.370</td>
</tr>
<tr>
<td valign="top" align="left">40</td>
<td valign="top" align="left">13</td>
<td valign="top" align="left">0.938</td>
<td valign="top" align="left">0.291</td>
<td valign="top" align="left">0.929</td>
<td valign="top" align="left">0.314</td>
</tr>
<tr>
<td valign="top" align="left">50</td>
<td valign="top" align="left">20</td>
<td valign="top" align="left">0.961</td>
<td valign="top" align="left">0.232</td>
<td valign="top" align="left">0.932</td>
<td valign="top" align="left">0.306</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_3_3">
<label>3.3.3</label>
<title>Effective wavelength selection by combination algorithm</title>
<p>Different wavelength selection methods have their own characteristics, which can be used by combination way to improve the effect of feature selection. At the same time, the synergistic effect of multiple methods can more comprehensively assess the importance of features, so as to screen out more representative spectral wavelengths (<xref ref-type="bibr" rid="B8">Gerretzen et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B30">Zhao et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B21">Shen et&#xa0;al., 2020</xref>). In this study, BVS-PLS algorithm was first used to select the spectral wavelengths and eliminate the non-information variables, and then SA algorithm was used to further reduce the multicollinearity between the variables (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). In order to evaluate the effectiveness of the algorithm, based on 59 wavelengths selected by BVS-PLS algorithm, the parameter settings were kept unchanged, and 20 spectral wavelengths were further selected by SA algorithm. The PLSR model was constructed based on the selected final wavelengths, and the model prediction results were shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. <inline-formula>
<mml:math display="inline" id="im59">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>= 0.935, <italic>RMSEC</italic> = 0.306%, <inline-formula>
<mml:math display="inline" id="im60">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>= 0.912, <italic>RMSEP</italic> = 0.354%, compared with the model constructed using only the wavelength selected by BVS-PLS algorithm, the accuracy of the prediction set was further improved. At the same time, compared with the PLSR prediction model constructed with 20 wavelengths selected by SA algorithm only, the wavelength selected by dual feature selection algorithm had better prediction performance for tomato SSC, and compared with the PLSR model constructed with 50 wavelengths selected by SA algorithm only, the prediction accuracy obtained by using dual feature selection algorithm only decreased a little, however, the number of wavelengths was dramatically reduced. In general, compared with 2048 wavelengths in the full-spectrum model, after grouping spectral wavelengths, BVS-PLS and SA screening, the PLSR model established using only 20 selected wavelengths (only 0.97% of the original spectrum) showed acceptable and robust prediction ability. This meant that the proposed wavelength selection technique can significantly simplify and improve the model performance, which was of great help to meet the demand of fast detection in the actual production line.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Flowchart of combinatorial feature selection algorithm.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1500819-g006.tif"/>
</fig>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Scatter plots of predicted SSC versus measured SSC for <bold>(A)</bold> calibration set and <bold>(B)</bold> prediction set based on PLSR model constructed with wavelengths selected by combination wavelength selection strategy.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-15-1500819-g007.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec id="s4" sec-type="conclusions">
<label>4</label>
<title>Conclusions</title>
<p>In this study, SSC of tomato was successfully measured under different acquisition orientation based on the developed multi-point Vis-NIR full transmission spectrum acquisition system. Four methods, MSC, SNV, GF and GF and MSC, were used to pretreat the original transmission spectrum of the tomato samples, and the Monte Carlo outlier detection method was employed to identify anomalous samples, while the effective wavelengths were determined using BVS-PLS, SA, and a combination of BVS-PLS and SA, respectively. Finally, PLSR linear model was established to predict tomato SSC. The results showed that the prediction performance of Z2 orientation was better than that of Z1 orientation, a shorter optical propagation path can build more stable model. Therefore, in the actual measurement, the orientation of the tomato should be placed according to Z2, so as to obtain the high-quality spectral data through the spectrometer. After abnormal sample removal and GF and MSC treatment, the spectral curve was smoother and the spectral scattering effect was suppressed. The feature selection algorithm combined with BVS-PLS and SA selected 20 effective spectral wavelengths from the original 2048 variables, and the prediction accuracy of <inline-formula>
<mml:math display="inline" id="im61">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>= 0.912 and <italic>RMSEP</italic> = 0.354% was obtained. In summary, the variable screening algorithm developed in this study can greatly reduce irrelevant information variables in the original spectral data while ensuring the accuracy of the model, thus eliminated multicollinearity in the spectral wavelengths and greatly improved the operating efficiency of the model. Next step of work, more tomato varieties and samples were used to improve and optimize the model performance for online application.</p>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>LC: Methodology, Visualization, Writing &#x2013; original draft. YZ: Investigation, Writing &#x2013; review &amp; editing. ZC: Validation, Writing &#x2013; review &amp; editing. RS: Software, Writing &#x2013; review &amp; editing. SL: Data curation, Writing &#x2013; review &amp; editing. JL: Funding acquisition, Project administration, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the Reform and Development Project of Beijing Academy of Agricultural and Forestry Sciences-Design and Control Implementation of Flexible Parallel Robot for Spherical Fruit and Vegetable Packaging, the Science and Technology Innovation Ability Construction Project of Beijing Academy of Agriculture and Forestry Science (Project No. KJCX20240503), the Construction of the Research and Innovation Platform of Beijing Academy of Agriculture and Forestry Science (PT2024-32) and the Outstanding Scientist Cultivation Project of Beijing Academy of Agriculture and Forestry Sciences (JKZX202405).</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec id="s9" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brito</surname> <given-names>A. A.</given-names>
</name>
<name>
<surname>de Campos</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Nascimento A. dos</surname> <given-names>R.</given-names>
</name>
<name>
<surname>de Corr&#xea;a</surname> <given-names>G. C.</given-names>
</name>
<name>
<surname>da Silva</surname> <given-names>F. A.</given-names>
</name>
<name>
<surname>Teixeira</surname> <given-names>D.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Determination of soluble solid content in market tomatoes using near-infrared spectroscopy</article-title>. <source>Food Control.</source> <volume>126</volume>, <elocation-id>108068</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.foodcont.2021.108068</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Shao</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>A variable selection method based on uninformative variable elimination for multivariate calibration of near-infrared spectra</article-title>. <source>Chemometrics Intelligent Lab. Systems.</source> <volume>90</volume>, <fpage>188</fpage>&#x2013;<lpage>194</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.chemolab.2007.10.001</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Improved Savitzky&#x2013;Golay-method-based fluorescence subtraction algorithm for rapid recovery of Raman spectra</article-title>. <source>Appl. Optics.</source> <volume>53</volume>, <elocation-id>5559</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1364/ao.53.005559</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dhanoa</surname> <given-names>M. S.</given-names>
</name>
<name>
<surname>L&#xf3;pez</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sanderson</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Lister</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Barnes</surname> <given-names>R. J.</given-names>
</name>
<name>
<surname>Ellis</surname> <given-names>J. L.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Methodology adjusting for least squares regression slope in the application of multiplicative scatter correction to near-infrared spectra of forage feed samples</article-title>. <source>J. Chemometrics</source> <volume>37</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/cem.3511</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Diniz</surname> <given-names>P. H. G. D.</given-names>
</name>
<name>
<surname>Pistonesi</surname> <given-names>M. F.</given-names>
</name>
<name>
<surname>Ara&#xfa;jo</surname> <given-names>M. C. U.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Using iSPA-PLS and NIR spectroscopy for the determination of total polyphenols and moisture in commercial tea samples</article-title>. <source>Analytical Methods</source> <volume>7</volume>, <fpage>3379</fpage>&#x2013;<lpage>3384</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1039/c4ay03099k</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ekinci</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Izci</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Yilmaz</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Simulated annealing aided artificial hummingbird optimizer for infinite impulse response system identification</article-title>. <source>IEEE Access.</source> <volume>11</volume>, <fpage>88627</fpage>&#x2013;<lpage>88636</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/access.2023.3303328</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fern&#xe1;ndez Pierna</surname> <given-names>J. A.</given-names>
</name>
<name>
<surname>Abbas</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Baeten</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Dardenne</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A Backward Variable Selection method for PLS regression (BVSPLS)</article-title>. <source>Analytica Chimica Acta</source> <volume>642</volume>, <fpage>89</fpage>&#x2013;<lpage>93</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aca.2008.12.002</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gerretzen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Szyma&#x144;ska</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Jansen</surname> <given-names>J. J.</given-names>
</name>
<name>
<surname>Bart</surname> <given-names>J.</given-names>
</name>
<name>
<surname>van Manen</surname> <given-names>H.-J.</given-names>
</name>
<name>
<surname>van den Heuvel</surname> <given-names>E. R.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>Simple and effective way for data preprocessing selection based on design of experiments</article-title>. <source>Analytical Chem.</source> <volume>87</volume>, <fpage>12096</fpage>&#x2013;<lpage>12103</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acs.analchem.5b02832</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Assessment of tomato soluble solids content and pH by spatially-resolved and conventional Vis/NIR spectroscopy</article-title>. <source>J. Food Engineering.</source> <volume>236</volume>, <fpage>19</fpage>&#x2013;<lpage>28</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jfoodeng.2018.05.008</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Inoue</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Sakaiya</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Takahashi</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Diagnostic mapping of canopy nitrogen content in rice based on hyperspectral measurements</article-title>. <source>Remote Sens. Environment.</source> <volume>126</volume>, <fpage>210</fpage>&#x2013;<lpage>221</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rse.2012.08.026</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lepot</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Aubin</surname> <given-names>J.-B.</given-names>
</name>
<name>
<surname>Clemens</surname> <given-names>F. H. L. R.</given-names>
</name>
<name>
<surname>Ma&#x161;i&#x107;</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Outlier detection in UV/Vis spectrophotometric data</article-title>. <source>Urban Water J.</source> <volume>14</volume>, <fpage>908</fpage>&#x2013;<lpage>921</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/1573062x.2017.1280515</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Detection of early decay in navel oranges by structured-illumination reflectance imaging combined with image enhancement and segmentation</article-title>. <source>Postharvest Biol. Technology.</source> <volume>196</volume>, <elocation-id>112162</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.postharvbio.2022.112162</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Identification of early decayed oranges using structured-illumination reflectance imaging coupled with fast demodulation and improved image processing algorithms</article-title>. <source>Postharvest Biol. Technology.</source> <volume>207</volume>, <elocation-id>112627</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.postharvbio.2023.112627</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Comparison and optimization of models for determination of sugar content in pear by portable vis-NIR spectroscopy coupled with wavelength selection algorithm</article-title>. <source>Food Analytical Methods</source> <volume>12</volume>, <fpage>12</fpage>&#x2013;<lpage>22</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s12161-018-1326-7</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Multi-agent simulated annealing algorithm with parallel adaptive multiple sampling for protein structure prediction in AB off-lattice model</article-title>. <source>Appl. Soft Computing.</source> <volume>62</volume>, <fpage>491</fpage>&#x2013;<lpage>503</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.asoc.2017.09.037</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>He</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Determination of effective wavelengths for discrimination of fruit vinegars using near infrared spectroscopy and multivariate analysis</article-title>. <source>Analytica Chimica Acta</source> <volume>615</volume>, <fpage>10</fpage>&#x2013;<lpage>17</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aca.2008.03.030</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhan</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Spectrum classification of citrus tissues infected by fungi and multispectral image identification of early rotten oranges</article-title>. <source>Spectrochimica Acta Part A: Mol. Biomolecular Spectroscopy.</source> <volume>279</volume>, <elocation-id>121412</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.saa.2022.121412</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mei</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An overview on optical non-destructive detection of bruises in fruit: Technology, method, application, challenge and trend</article-title>. <source>Comput. Electron. Agriculture.</source> <volume>213</volume>, <elocation-id>108195</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2023.108195</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ning</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Quantitative detection of zearalenone in wheat grains based on near-infrared spectroscopy</article-title>. <source>Spectrochimica Acta Part A: Mol. Biomolecular Spectroscopy.</source> <volume>280</volume>, <elocation-id>121545</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.saa.2022.121545</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ozturkoglu-Budak</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Aksahin</surname> <given-names>I.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Multivariate characterization of fresh tomatoes and tomato-based products based on mineral contents including major trace elements and heavy metals</article-title>. <source>J. Food Nutr. Res.</source> <volume>55</volume>, <fpage>214</fpage>&#x2013;<lpage>221</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5072/ZENODO.14744</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shen</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.-Z.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Discrimination of gentiana and its related species using IR spectroscopy combined with feature selection and stacked generalization</article-title>. <source>Molecules.</source> <volume>25</volume>, <elocation-id>1442</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/molecules25061442</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Karimi</surname> <given-names>H. R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Coverage path planning for cleaning robot based on improved simulated annealing algorithm and ant colony algorithm</article-title>. <source>Signal Image Video Processing.</source> <volume>18</volume>, <fpage>3275</fpage>&#x2013;<lpage>3284</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11760-023-02989-y</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Study on outlier detection method of the near infrared spectroscopy analysis by probability metric</article-title>. <source>Spectrochimica Acta Part A: Mol. Biomolecular Spectroscopy.</source> <volume>280</volume>, <elocation-id>121473</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.saa.2022.121473</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Suman</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>A survey of simulated annealing as a tool for single and multiobjective optimization</article-title>. <source>J. Operational Res. Society.</source> <volume>57</volume>, <fpage>1143</fpage>&#x2013;<lpage>1160</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1057/palgrave.jors.2602068</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname> <given-names>B.</given-names>
</name>
<name>
<surname>You</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>An intelligent near-infrared diffuse reflectance spectroscopy scheme for the non-destructive testing of the sugar content in cherry tomato fruit</article-title>. <source>Electronics.</source> <volume>11</volume>, <elocation-id>3504</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/electronics112135043</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Optimization and compensation of models on tomato soluble solids content assessment with online Vis/NIRS diffuse transmission system</article-title>. <source>Infrared Phys. Technol</source>. <volume>121</volume>, <elocation-id>104050</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.infrared.2022.104050</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Nondestructive evaluation of soluble solids content in tomato with different stage by using Vis/NIR technology and multivariate algorithms</article-title>. <source>Spectrochimica Acta Part A: Mol. Biomolecular Spectroscopy.</source> <volume>248</volume>, <elocation-id>119139</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.saa.2020.119139</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhan</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Determination of soluble solids content in oranges using visible and near infrared full transmittance hyperspectral imaging with comparative analysis of models</article-title>. <source>Postharvest Biol. Technology.</source> <volume>163</volume>, <elocation-id>111148</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.postharvbio.2020.111148</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Adaptive gaussian filter based on ICEEMDAN applying in non-gaussian non-stationary noise</article-title>. <source>Circuits Systems Signal Processing.</source> <volume>43</volume>, <fpage>4272</fpage>&#x2013;<lpage>4297</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00034-024-02642-0</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Pharmaceutical analysis model robustness from bagging-pls and pls using systematic tracking mapping</article-title>. <source>Front. Chem.</source> <volume>6</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fchem.2018.00262</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Improving SSC detection accuracy of cherry tomatoes by feature synergy and complementary spectral bands combination</article-title>. <source>Postharvest Biol. Technology.</source> <volume>213</volume>, <elocation-id>112922</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.postharvbio.2024.112922</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>