<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Earth Sci.</journal-id>
<journal-title>Frontiers in Earth Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Earth Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-6463</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1106799</article-id>
<article-id pub-id-type="doi">10.3389/feart.2022.1106799</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Earth Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>TOC interpretation of lithofacies-based categorical regression model: A case study of the Yanchang formation shale in the Ordos basin, NW China</article-title>
<alt-title alt-title-type="left-running-head">Yin et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/feart.2022.1106799">10.3389/feart.2022.1106799</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Yin</surname>
<given-names>Jintao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2043328/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gao</surname>
<given-names>Chao</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2050760/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Cheng</surname>
<given-names>Ming</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1800612/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liang</surname>
<given-names>Quansheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xue</surname>
<given-names>Pei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hao</surname>
<given-names>Shiyan</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Qianping</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Shaanxi Yanchang Petroleum (Group) Corp Ltd.</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Shaanxi Key Laboratory of Lacustrine Shale Gas Accumulation and Exploitation</institution>, <addr-line>Xi&#x2019;an</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Institute of Geology and Geophysics</institution>, <institution>Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/942163/overview">Qingqiang Meng</ext-link>, SINOPEC Petroleum Exploration and Production Research Institute, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1819466/overview">Guo Xiaobo</ext-link>, Xi&#x2019;an Shiyou University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1779341/overview">Yiming Yan</ext-link>, China University of Petroleum (East China), China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Ming Cheng, <email>chengming@mail.iggcas.ac.cn</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Geochemistry, a section of the journal Frontiers in Earth Science</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>20</day>
<month>01</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>10</volume>
<elocation-id>1106799</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>11</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>27</day>
<month>12</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Yin, Gao, Cheng, Liang, Xue, Hao and Zhao.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Yin, Gao, Cheng, Liang, Xue, Hao and Zhao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>In this paper, taking the shale of Chang 7-Chang 9 oil formation in Yanchang Formation in the southeastern Ordos Basin as an example, through the study of shale heterogeneity characteristics, starting from the preprocessing of supervision data set, a logging interpretation method of total organic carbon content (TOC) on the lithofacies-based Categorical regression model (LBCRM) is proposed. It is show that: 1) Based on core observation, and Differences of sedimentation and structure, five lithofacies developed in the Yanchang Formation: shale shale facies, siltstone/ultrafine sandstone facies, tuff facies, argillaceous shale facies with silty lamina and argillaceous shale facies with tuff lamina. 2) The strong heterogeneity of shale makes it difficult to accurately explain the TOC distribution of shale intervals in the application of model-based interpretation methods. The LBCRM interpretation method based on the understanding of shale heterogeneity can effectively reduce the influence of formation factors other than TOC on the prediction accuracy by studying the characteristics of shale heterogeneity and constructing a TOC interpretation model for each lithofacies category. At the same time, the degree of unbalanced distribution of data is reduced, so that the data mining algorithm achieves better prediction effect. 3) The interpretability of lithofacies logging ensures the wellsite application based on the classification and regression model of lithofacies. Compared with the traditional homogeneous regression model, the prediction performance has been greatly improved, TOC segment prediction is more accurate. 4) The LBCRM method based on shale heterogeneity can better understand the reasons for the deviation of the traditional model-based interpretation method. After being combined with the latter, it can make logging data provide more useful information.</p>
</abstract>
<kwd-group>
<kwd>ordos basin</kwd>
<kwd>Yan&#x2019;an area</kwd>
<kwd>lacustrine oil shale</kwd>
<kwd>lithofacies classification regression</kwd>
<kwd>TOC interpretation model</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Organic matter content is an indispensable basic data for source rock evaluation, shale oil and gas reservoir evaluation and sweet spot prediction. (<xref ref-type="bibr" rid="B13">Curtis, 2002</xref>; <xref ref-type="bibr" rid="B30">Passey et al., 2010</xref>; <xref ref-type="bibr" rid="B37">Sondergeld et al., 2010</xref>; <xref ref-type="bibr" rid="B3">Alfred and Vernik, 2012</xref>; <xref ref-type="bibr" rid="B25">Ma, 2015</xref>; <xref ref-type="bibr" rid="B4">Altowairqi et al., 2015</xref>; <xref ref-type="bibr" rid="B2">Aldrich and seidle, 2018</xref>; <xref ref-type="bibr" rid="B16">Guo et al., 2021</xref>; <xref ref-type="bibr" rid="B44">Wei et al., 2021</xref>; <xref ref-type="bibr" rid="B28">Meng, 2022</xref>). Laboratory core test and analysis technology is the most direct and accurate means to obtain the organic matter content of shale, in which total organic carbon content (TOC) is the most readily available and commonly used characterization index of organic matter content. Restricted by the lack of core data or incomplete coring in most wells, the interpretation of formation TOC with high resolution and high coverage logging data is an important means for rapid, accurate and continuous quantitative evaluation of organic matter content in shale formations (<xref ref-type="bibr" rid="B46">Yu et al., 2017</xref>; <xref ref-type="bibr" rid="B42">Wang et al., 2019</xref>; <xref ref-type="bibr" rid="B21">Liang et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Chan et al., 2022</xref>; <xref ref-type="bibr" rid="B27">Meng et al., 2022</xref>; <xref ref-type="bibr" rid="B49">Zhao et al., 2022</xref>).</p>
<p>At present, a large number of TOC logging interpretation methods, techniques or models have been proposed. These methods can be divided into two categories: model-driven and data-driven (<xref ref-type="bibr" rid="B18">Huang and Williamson, 1996</xref>). Model-driven methods include formation density curve method (<xref ref-type="bibr" rid="B34">Schmoker, 1979</xref>; <xref ref-type="bibr" rid="B35">Schmoker and Hester, 1981</xref>), natural gamma intensity method (<xref ref-type="bibr" rid="B35">Schmoker, 1981</xref>; fertl and Chilinger, 1988), I-x method (<xref ref-type="bibr" rid="B14">Dellenbach et al., 1983</xref>), &#x394;logR and its improved method (<xref ref-type="bibr" rid="B31">Passey et al., 1990</xref>; <xref ref-type="bibr" rid="B43">wang et al., 2016</xref>; <xref ref-type="bibr" rid="B50">zhao et al., 2017</xref>), CARBOLOG (<xref ref-type="bibr" rid="B9">Carpentier et al., 1991</xref>), etc. This type of method constructs a statistical relationship between logging response and TOC through specific assumptions (<xref ref-type="bibr" rid="B37">Sondergeld et al., 2010</xref>). For example, the formation density curve and the natural gamma intensity method construct the TOC logging interpretation method through the linear volume equation of the logging response (<xref ref-type="bibr" rid="B18">Huang and Williamson, 1996</xref>), and the &#x394;logR establishes the non-linear relationship between the &#x394;logR and the TOC by obtaining the superposition baseline of the porosity curve and the resistivity curve at the pure water-bearing non-hydrocarbon source rock under the premise of the known shale mature section (<xref ref-type="bibr" rid="B31">Passey et al., 1990</xref>; <xref ref-type="bibr" rid="B30">2010</xref>).</p>
<p>
<xref ref-type="bibr" rid="B18">Huang and Williamson (1996)</xref> pointed out that the model-driven method need to determine the key parameter to accurately estimate the organic matter content of the shale section. The above drawbacks restrict the application of model-driven methods in the interpretation of organic matter content and promote the development of data-driven methods (<xref ref-type="bibr" rid="B18">Huang and Williamson, 1996</xref>). Different from the model-driven method, the data-driven method can fully explore the statistical relationship between multi-logging response characteristics and TOC, which is more suitable for TOC interpretation of strongly heterogeneous shale (<xref ref-type="bibr" rid="B18">Huang and Williamson, 1996</xref>). Currently, a large number of data mining algorithms have been applied to TOC logging interpretation, including multiple linear regression, Gaussian mixture, optimization algorithm, SVM, BP neural network, deep neural network, etc., (Mendelzon and Roksoz, 1985; <xref ref-type="bibr" rid="B18">Huang and Williamson, 1996</xref>; <xref ref-type="bibr" rid="B40">Wang et al., 2014</xref>; <xref ref-type="bibr" rid="B38">Tan et al., 2015</xref>; <xref ref-type="bibr" rid="B46">Yu et al., 2017</xref>; <xref ref-type="bibr" rid="B53">Zhu et al., 2020</xref>; <xref ref-type="bibr" rid="B52">Zheng et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Chan et al., 2022</xref>).</p>
<p>In the data-driven TOC interpretation technology, there are two challenges: First, the formation logging response is not only affected by TOC, but also by multi-formation factors such as particle size, mineral composition, element composition, pore development degree, pore fluid properties, etc., resulting in the logging response and organic matter content is not a simple linear relationship (<xref ref-type="bibr" rid="B18">Huang et al., 1996</xref>; yang et al., 2004; rezaee et al., 2007). The above characteristics have caused a prominent problem, whether the conventional logging series can provide sufficient features to make the TOC interpretation have high enough accuracy, in other words, in the formation with the same or similar logging response, whether the samples have different TOC values. <xref ref-type="bibr" rid="B10">Chan et al. (2020)</xref> showed that the accuracy of TOC interpretation based solely on conventional logging series may not be ideal. The TOC deep learning interpretation model constructed by adding element information to conventional logging series data is significantly better than the results of <xref ref-type="bibr" rid="B26">Mahmoud et al. (2017)</xref> that rely solely on conventional logging prediction models (<xref ref-type="bibr" rid="B10">Chan et al., 2020</xref>). It can be seen that the simple introduction of more complex machine learning algorithms cannot completely solve the accurate interpretation of TOC. It is also necessary to understand the above problems from the perspective of data characteristics, which is particularly important in shale oil and gas reservoirs with strong heterogeneity of lithology, mineral composition and elemental composition.</p>
<p>Another problem comes from the data mining algorithm itself. In all data-driven TOC interpretation methods, the goal is to minimize the difference between the predicted value and the true value of the expected value (such as MSE and RMSE, etc.) (<xref ref-type="bibr" rid="B18">Huang and Williamson, 1996</xref>; <xref ref-type="bibr" rid="B40">Wang et al., 2014</xref>; <xref ref-type="bibr" rid="B38">Tan et al., 2015</xref>; <xref ref-type="bibr" rid="B47">Yu et al., 2017</xref>; <xref ref-type="bibr" rid="B53">Zhu et al., 2020</xref>; <xref ref-type="bibr" rid="B52">Zheng et al., 2021</xref>; <xref ref-type="bibr" rid="B10">Chan et al., 2022</xref>), which is the most direct indicator of learning algorithms in model training and performance verification. However, TOC test samples are often sampled by equidistant or random methods. The strong heterogeneity of shale inevitably causes some TOC numerical interval samples to be more concentrated. The TOC data exhibit skewed distribution with a long tail (Yu et al., 2019; <xref ref-type="bibr" rid="B41">Wang et al., 2012</xref>), causing an imbalance in data distribution (<xref ref-type="bibr" rid="B6">Branco et al., 2016</xref>). The learning goal of minimizing the expected difference makes the learning algorithm pay more attention to the characteristics of high-frequency distribution samples, resulting in lower prediction accuracy for data with a small number of samples (<xref ref-type="bibr" rid="B6">Branco et al., 2016</xref>; <xref ref-type="bibr" rid="B5">2018</xref>). Unfortunately, the TOC interval with low data density may be the focus of shale reservoir research, such as shale sections with high TOC distribution. At present, the application of learning algorithms in imbalanced data is still less involved in regression problems such as TOC logging interpretation (<xref ref-type="bibr" rid="B6">Branco et al., 2016</xref>; <xref ref-type="bibr" rid="B5">2018</xref>).</p>
<p>In view of the above problems, this paper takes Yanchang Formation in Ordos Basin as the research object, and proposes a logging interpretation method of organic carbon content based on rock facies classification regression model (LBCRM) from the preprocessing of supervised data sets. This method adds an additional dimension of lithofacies to the TOC-logging response monitoring data set through the study of shale heterogeneity characteristics. The TOC interpretation sub-model based on SVM algorithm is constructed by classification, which effectively reduces the influence of formation factors other than TOC on TOC interpretation accuracy. At the same time, the degree of unbalanced data distribution is reduced, which makes the data mining algorithm achieve better prediction results. XGboost can be used to construct a high-precision rock facies logging identification method, which ensures the availability of rock facies and makes this method have practical application potential. In addition, based on the analysis of heterogeneity characteristics, the interpretation results of this method can also be combined with the traditional model-driven method to obtain more formation parameters.</p>
</sec>
<sec sec-type="materials" id="s2">
<title>2 Materials</title>
<p>This study is based on the Yanchang Formation shale in the southeastern Ordos Basin (<xref ref-type="fig" rid="F1">Figure 1A</xref>). The shale is a Triassic continental deposit, and the mud shale section is located in the Chang 7 &#x223c; Chang 9 oil formation. The data come from core samples and conventional logging curves of 12 wells (<xref ref-type="fig" rid="F1">Figure 1B</xref>). As shown in <xref ref-type="table" rid="T1">Table 1</xref>, based on the core description of the above 12 wells, the samples were selected for TOC, mineral composition, extraction and pyrolysis test, and the core homing work was carried out.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>
<bold>(A)</bold> Location of Ordos Basin and study area (modified by Yang et al., 2005); <bold>(B)</bold> Horizontal distribution of wells in the study area; <bold>(C)</bold> TOC frequency distribution histogram.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g001.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Testing data and conventional well logs used in this study.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Well name</th>
<th colspan="3" align="center">Testing</th>
<th rowspan="2" align="center">Gamma ray (GR)</th>
<th rowspan="2" align="center">Sonic (DT)</th>
<th rowspan="2" align="center">Resistivity (ILD,ILM, Rt)</th>
<th rowspan="2" align="center">Density (DEN)</th>
<th rowspan="2" align="center">SGR (URAN, THOR, POTA)</th>
<th rowspan="2" align="center">Neutron porosity (CNL)</th>
<th rowspan="2" align="center">Caliper (CAL)</th>
</tr>
<tr>
<th align="center">TOC</th>
<th align="center">Mineral composition</th>
<th align="center">Pyrolysis</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">YY2</td>
<td align="center">16</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">YY12</td>
<td align="center">25</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">YY18</td>
<td align="center">50</td>
<td align="center">25</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">YY22</td>
<td align="center">104</td>
<td align="center">52</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">YY27</td>
<td align="center">25</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">27</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">YY28</td>
<td align="center">52</td>
<td align="center">35</td>
<td align="center">26</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">FY1</td>
<td align="center">74</td>
<td align="center">21</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">FY3</td>
<td align="center">30</td>
<td align="center">23</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">B36</td>
<td align="center">20</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">WY1</td>
<td align="center">46</td>
<td align="center">13</td>
<td align="center">12</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
</tr>
<tr>
<td align="center">W169</td>
<td align="center">8</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
</tr>
<tr>
<td align="center">DT5</td>
<td align="center">9</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#x221a;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
<td align="center">
<bold>&#xd7;</bold>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Among them, the TOC sample size is 459. Statistics show that the TOC distribution is 0.34&#xa0;wt%&#x223c;29.11&#xa0;wt% (4.76% on average). From <xref ref-type="fig" rid="F1">Figure 1C</xref>, it can be found that the data exhibit skewed distribution with a long tail. The high-density data distribution area is located at 3&#xa0;wt% &#x223c; 8&#xa0;wt%, showing that the data has an unbalanced distribution (<xref ref-type="bibr" rid="B7">Buda et al., 2018</xref>; <xref ref-type="bibr" rid="B22">Liu et al., 2019</xref>).</p>
<p>In addition to the TOC test, the whole rock mineral composition and pyrolysis test were also carried out in this study. These data were used to illustrate the differences in mineral composition and oil content of different lithofacies.</p>
<p>Except W169 and DT5 wells which lack Density, SGR and Neutron logging series, 10 wells have complete logging series. In this study, the 10 wells were selected to construct the LBCRM method, W169 and DT5 were used for the extended application of the LBCRM method.</p>
</sec>
<sec sec-type="methods" id="s3">
<title>3 Methodology</title>
<sec id="s3-1">
<title>3.1 Principle of LBCRM</title>
<p>In essence, data-driven TOC logging interpretation is a typical data regression problem based on learning algorithms. Suppose that a supervised data set <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>x</italic>&#x2208;<italic>X</italic>, <italic>y</italic>&#x2208;<italic>Y</italic>, is derived from the joint distribution <italic>P</italic>
<sub>
<italic>X</italic> &#xd7; <italic>Y</italic>
</sub>. The goal of the data-driven method is to establish a mapping relationship <italic>f</italic>&#x2208;<italic>F</italic>:<italic>X</italic> -&#x3e;<italic>Y</italic>, such that the expected error <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>Y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is minimized, where <italic>L</italic> (<italic>f</italic> (<italic>x</italic>), <italic>y</italic>) is the loss function, representing the difference between the predicted value <italic>f</italic> (<italic>x</italic>) and the supervised target <italic>y</italic> value. In practice, the joint distribution <italic>P</italic>
<sub>
<italic>X</italic> &#xd7; <italic>Y</italic>
</sub> is unknown, <italic>x</italic> and <italic>y</italic> generally take values from the supervised data set <italic>D</italic>
<sub>
<italic>t</italic>
</sub>, so the objective of the regression problem is to minimize <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. When the supervised data set is large enough, <italic>&#x3b5; &#x3d;</italic> &#x7c;<italic>&#x3b5;</italic>
<sub>
<italic>ex</italic>
</sub>
<italic>-&#x3b5;</italic>
<sub>
<italic>em</italic>
</sub>&#x7c; is small enough, so that the regression fitting relationship <italic>f</italic> has better prediction effect. For logging interpretation, <italic>x</italic> is the conventional logging response, <italic>f</italic> is the formation characteristic parameters, including mineral composition, element composition and organic matter content.</p>
<p>Compared with the easily available TOC data, other formation parameter data are often difficult to obtain for various reasons. Therefore, the target output in the supervised data set <italic>D</italic>
<sub>
<italic>t</italic>
</sub> of TOC logging interpretation is only TOC data. This requires that conventional logging responses can provide sufficient differentiated features to distinguish TOC values. A comparative study by Chan et al. (2020) and Mahmoud et al. (2017) found that prediction accuracy can be significantly improved by adding dimensional information to conventional logging responses, suggesting that conventional logging responses may not be sufficient to provide complete features for accurate interpretation of TOC.</p>
<p>Similar to <xref ref-type="bibr" rid="B12">Chan et al. (2020)</xref>, the TOC interpretation model based on rock facies classification and regression improves the prediction accuracy of TOC by adding additional dimension information to logging information. Based on the study of shale heterogeneity, this method constructs a relatively homogeneous lithofacies unit and uses it as additional information to constrain TOC interpretation. The mathematical expression of the regression target of this method is to divide the <italic>D</italic>
<sub>
<italic>t</italic>
</sub> data set into <italic>m</italic> subsets <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msubsup>
<mml:mi>D</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, and establish a function mapping relationship <italic>f</italic>
<sub>
<italic>j</italic>
</sub> for each subset to minimize Eq. <xref ref-type="disp-formula" rid="e1">1</xref>:<disp-formula id="e1">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msubsup>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mi>j</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:msubsup>
<mml:mi>D</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>j</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <italic>m</italic> is the number of types of lithofacies units,<italic>j</italic>&#x2208; [1.2,&#x2026;m].</p>
<p>
<xref ref-type="fig" rid="F2">Figure 2</xref> shows the basic idea of this method. Traditional data-driven TOC interpretation methods use a uniform regression model (URM) when constructing prediction models. As shown in <xref ref-type="fig" rid="F2">Figure 2A</xref>, firstly, the homogeneous regression model ignores that the input data is not enough to provide enough differentiated features to describe the output target. Secondly, the data imbalance in the supervised data makes the learning algorithm have the data characteristics in the rectangular area in <xref ref-type="fig" rid="F2">Figure 2B</xref>, but the fitting model in <xref ref-type="fig" rid="F2">Figure 2A</xref> cannot have good prediction performance for the data outside the gray rectangular area. The classification fitting regression model shown in <xref ref-type="fig" rid="F2">Figure 2C</xref> can increase the type dimension information, so that the learning algorithm can obtain a more accurate prediction model in the data within different categories. At the same time, as shown in <xref ref-type="fig" rid="F2">Figure 2D</xref>, this method can also reduce the imbalance of the data, so that the learning algorithm will not only focus on the data with high frequency distribution, especially for the high density of local data distribution caused by the coincidence of different types of data.</p>
<fig id="F2" position="float">
<label>FIGURE. 2</label>
<caption>
<p>Schematic diagram of regression prediction model based on rock facies classification <bold>(A)</bold> The effect of using a uniform regression fitting model in the case of incomplete input data and unbalanced supervised data; <bold>(B)</bold> The unbalanced distribution characteristics of the data set; <bold>(C)</bold> The effect of using classification fitting regression fitting model in the case of incomplete input data and unbalanced supervised data; <bold>(D)</bold> classification fitting regression subdataset imbalance distribution reduction.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g002.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Classification of lithofacies</title>
<p>At present, the classification of shale rock facies is mainly divided into two categories. One is based on the difference of sedimentary and structure on the core scale of mud shale section (<xref ref-type="bibr" rid="B36">Singh et al., 2008</xref>; <xref ref-type="bibr" rid="B51">Zhen et al., 2016</xref>; Kristen, 2015; <xref ref-type="bibr" rid="B23">Long et al., 2022</xref>; <xref ref-type="bibr" rid="B48">Zhang et al., 2022</xref>); the second is based on rock physics parameters, especially mineral composition parameters (<xref ref-type="bibr" rid="B41">Wang et al., 2012</xref>; <xref ref-type="bibr" rid="B15">Gao et al., 2018</xref>; <xref ref-type="bibr" rid="B29">Ou et al., 2018</xref>; <xref ref-type="bibr" rid="B33">Schlanser, 2015</xref>).</p>
<p>In this study, the rock has three corresponding characteristics. One is that the rock facies type is not easy to be too complicated for the consideration of well site application, which makes it difficult to establish a logging identification method with high prediction accuracy. The second is easy to obtain. On the one hand, it is conducive to the formation of large-scale data sets, on the other hand, the rock type classification using TOC data; The third is the thickness of rock facies should be above the vertical resolution of the logging. Taking DEN with the highest vertical resolution in conventional logging as an example, the thickness of rock facies should be at least 30&#xa0;cm.</p>
<p>Due to the numerous petrophysical parameters affecting the logging response, a more complex classification scheme will be formed in the rock facies construction, and it is easy to fall into the rock facies classification only for TOC data with petrophysical parameters. Therefore, this study uses the difference of sedimentary and structure on the core scale as the basis for the division of rock facies. At the same time, in order to avoid the occurrence of complex lithofacies types, only the sedimentary characteristics that have obvious influence on rock physics characteristics are considered. In addition, in order to correspond to the vertical resolution of logging, the thickness of a single rock facies layer is at least 30&#xa0;cm.</p>
</sec>
<sec id="s3-3">
<title>3.3 Machine learning method</title>
<p>The data mining algorithms used in this study include SVR (support vector machine) and XGboost. In addition, genetic algorithm is used to optimize the hyperparameters of the above two algorithms, and K-fold cross validation is used to improve the generalization ability of the training model.</p>
<sec id="s3-3-1">
<title>3.3.1 SVR method</title>
<p>SVR has incomparable advantages in data mining of small sample data sets. Considering that there may be a small amount of data in some data sets after the construction of sub-data sets, this paper chooses SVR as the basic data mining algorithm for TOC logging interpretation. The basic concept of SVR method is to project the input data into a higher dimension by kernel function, so as to find a hyperplane to establish a regression function. For a given data set {(<italic>x</italic>
<sub>
<italic>1</italic>
</sub>,<italic>y</italic>
<sub>
<italic>1</italic>
</sub>),&#x2026;&#x2026;, (<italic>x</italic>
<sub>
<italic>l</italic>
</sub>,<italic>y</italic>
<sub>
<italic>l</italic>
</sub>) }, where x<sub>
<italic>i</italic>
</sub>&#x2208;<italic>R</italic>
<sup>n</sup> is the input data, y<sub>
<italic>i</italic>
</sub>&#x2208;<italic>R</italic>
<sup>1</sup> is the target output value, and the SVR estimation function is:<disp-formula id="e2">
<mml:math id="m6">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mi>w</mml:mi>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:msup>
<mml:mo>&#xb7;</mml:mo>
<mml:mi mathvariant="normal">&#x3a6;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="normal">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <italic>w</italic> and <italic>b</italic> are hyperplane parameters, &#x3a6;(<italic>x</italic>) denotes the eigenvectors after <italic>x</italic> projection. The standard form of SVR for solving hyperplane parameters is (<xref ref-type="bibr" rid="B39">Vapnik, 1998</xref>):<disp-formula id="e3a">
<mml:math id="m7">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mi mathvariant="bold">min</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msup>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mi>w</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>w</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(3a)</label>
</disp-formula>
</p>
<p>&#x53d7;&#x5236;&#x4e8e;.</p>
<p>Subject to<disp-formula id="e3b">
<mml:math id="m8">
<mml:mrow>
<mml:msup>
<mml:mi>w</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>&#xb7;</mml:mo>
<mml:mi mathvariant="normal">&#x3a6;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3b)</label>
</disp-formula>
<disp-formula id="equ1">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>w</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>&#xb7;</mml:mo>
<mml:mi mathvariant="normal">&#x3a6;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>b</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="equ2">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msubsup>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <italic>C</italic> is the penalty coefficient or regularization parameter, and <italic>&#x3b5;</italic>, <italic>&#x3be;</italic>,<italic>&#x3be;</italic>&#x2a;&#x2208;R are slack variables introduced to penalize the fitting function. Eq. <xref ref-type="disp-formula" rid="e1">1</xref> can be transformed into a dual problem to solve, and the original problem is transformed into its corresponding Lagrangian function form, and by minimizing:<disp-formula id="e4a">
<mml:math id="m11">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:mi mathvariant="bold">min</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4a)</label>
</disp-formula>
</p>
<p>Subject to<disp-formula id="e4b">
<mml:math id="m12">
<mml:mrow>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>l</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4b)</label>
</disp-formula>
<disp-formula id="equ3">
<mml:math id="m13">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>l</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <italic>&#x3b1;</italic>
<sub>
<italic>i</italic>
</sub> &#x3d; (<italic>&#x3b1;</italic>
<sub>
<italic>1</italic>
</sub>, <italic>&#x3b1;</italic>
<sub>
<italic>2</italic>
</sub>,&#x2026;<italic>&#x3b1;</italic>
<sub>
<italic>l</italic>
</sub>) is the Lagrange multiplier, and <italic>K</italic> (<italic>x</italic>
<sub>
<italic>i</italic>
</sub>,<italic>x</italic>
<sub>
<italic>j</italic>
</sub>) is the kernel function. The final regression equation is:<disp-formula id="e5">
<mml:math id="m14">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>K</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>In this paper, polynomial kernel function, radial basis kernel function (RBF) and sigmod kernel function are selected to explain TOC respectively, so as to optimize the best kernel function type.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 XGboost</title>
<p>XGboost was first proposed by <xref ref-type="bibr" rid="B11">Chen and Guestrin (2016)</xref>. It is a machine learning algorithm that relies on the boosting principle and explores weak learners to comprehensively predict. This is mainly due to its well-known high prediction accuracy. Its basic principle is to generate a sub-classifier to fit the prediction residuals of the previous sub-classifiers, thereby continuously reducing the residuals between the true value and the predicted value, and finally integrating all sub-classifiers to give the final prediction result. The expression is:<disp-formula id="e6">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>K</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>Among them, <inline-formula id="inf5">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the calculated predictive value of the <italic>i</italic> th sample; <italic>K</italic> denotes the number of decision trees; <italic>f</italic>
<sub>
<italic>k</italic>
</sub> denotes the <italic>k</italic>th submodel; <italic>x</italic>
<sub>
<italic>i</italic>
</sub> represents the input feature of the <italic>i</italic>th sample; <italic>F</italic> represents the set of sub-classifiers. In the Xgboost sub-classifier, the Classification and Regression Tree is usually selected. In the Xgboost algorithm, the objective function is composed of a loss function and regularization parameters. The expression is:<disp-formula id="e7">
<mml:math id="m17">
<mml:mrow>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3c6;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mi>k</mml:mi>
</mml:munder>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>&#x3bb;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>Among them, <inline-formula id="inf6">
<mml:math id="m18">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the residual between the predicted value <inline-formula id="inf7">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the target value <italic>y</italic>
<sub>
<italic>i</italic>
</sub>; <italic>f</italic>
<sub>
<italic>k</italic>
</sub> is the function expression of the <italic>k</italic> sub-classifier; <inline-formula id="inf8">
<mml:math id="m20">
<mml:mrow>
<mml:mi>&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the penalty term of the model complexity, which can be used to smooth the final learned weights to avoid overfitting. XGboost is trained iteratively to obtain an approximation of <italic>L</italic>(<italic>&#x3c6;</italic>). Assuming that the sub-classifier trained in the <italic>t</italic> iteration is <italic>f</italic>
<sub>
<italic>t</italic>
</sub>, after the <italic>t</italic> iteration, the objective function can be expressed as:<disp-formula id="e8">
<mml:math id="m21">
<mml:mrow>
<mml:msup>
<mml:mi>L</mml:mi>
<mml:mi>t</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>Equation. <xref ref-type="disp-formula" rid="e8">8</xref> can be further optimized using second-order approximation:<disp-formula id="e9">
<mml:math id="m22">
<mml:mrow>
<mml:msup>
<mml:mi>L</mml:mi>
<mml:mi>t</mml:mi>
</mml:msup>
<mml:mo>&#x2245;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where, <italic>g</italic>
<sub>
<italic>i</italic>
</sub> and <italic>h</italic>
<sub>
<italic>i</italic>
</sub> are the first-order and second-order partial derivatives (gradients) of <italic>l</italic>, respectively, where <inline-formula id="inf9">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mo>&#x2202;</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:msub>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> &#x548c; <inline-formula id="inf10">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mo>&#x2202;</mml:mo>
<mml:msup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mi>l</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msubsup>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. After taking them into Eq. <xref ref-type="disp-formula" rid="e9">9</xref> and removing the constant term, we can obtain:<disp-formula id="e10">
<mml:math id="m25">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>Define <italic>I</italic>
<sub>
<italic>j</italic>
</sub>&#x3d;{<italic>i</italic> &#x7c; <italic>q</italic> (<italic>x</italic>
<sub>
<italic>i</italic>
</sub>) &#x3d; <italic>j</italic>} as an instance set of leaf node <italic>j</italic>. Eq. <xref ref-type="disp-formula" rid="e10">10</xref> is rewritten by extending &#x3a9; to:<disp-formula id="e11">
<mml:math id="m26">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>&#x3bb;</mml:mi>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
<mml:msup>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msubsup>
<mml:mi>w</mml:mi>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>Therefore, the objective function is transformed into a function of the first and second partial derivatives of the loss function <italic>l</italic>, the leaf node weight, and the number of leaf nodes. In the case of fixed tree structure <italic>q</italic>(<italic>x</italic>) the optimal weight <italic>w</italic>
<sub>
<italic>j</italic>
</sub>&#x2a;of leaf node <italic>j</italic> can be calculated by the following formula:<disp-formula id="e12">
<mml:math id="m27">
<mml:mrow>
<mml:msubsup>
<mml:mi>w</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>&#x2a;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>The optimal solution formula of the objective function is as follows:<disp-formula id="e13">
<mml:math id="m28">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mo>&#x223c;</mml:mo>
</mml:mover>
<mml:mi>t</mml:mi>
</mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
<p>XGboost iteratively adds branches to construct sub-classifiers on the initial leaf nodes through a greedy algorithm to determine the optimal tree structure of the CART tree. Suppose there is a leaf node, <italic>I</italic>
<sub>
<italic>L</italic>
</sub> and <italic>I</italic>
<sub>
<italic>R</italic>
</sub> are instances of the left and right nodes after the node is branched. Let <italic>I</italic> &#x3d; <italic>I</italic>
<sub>
<italic>L</italic>
</sub>&#x222a;<italic>I</italic>
<sub>
<italic>R</italic>
</sub>, then the loss after branching is reduced to:<disp-formula id="e14">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mrow>
<mml:mrow>
<mml:munder>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
</p>
<p>If <italic>L</italic>
<sub>
<italic>split</italic>
</sub> is greater than 0, the objective function decreases after the leaf node is split into two leaf nodes, so as to determine the node segmentation. On this basis, XGboost is optimized by feature pre-ranking, quantile approximation, and parallel lookup to quickly find the nearest split point.</p>
</sec>
<sec id="s3-3-3">
<title>3.3.3 Genetic algorithm</title>
<p>In the SVM and XGboost algorithm, there are a large number of hyper-parameters, which will affect the final prediction results. Therefore, it is necessary to use hyper-parameter optimization algorithm to determine which hyper-parameter system the SVM and XGboost algorithm can achieve the best prediction results. This study used Genetic Algorithm to optimize hyperparameters.</p>
<p>Genetic algorithm was first proposed by <xref ref-type="bibr" rid="B17">Holland (1973)</xref>. It is a parallel stochastic optimization algorithm developed from the simulation of natural genetic mechanism and biological evolution theory. The genetic algorithm starts with a set of randomly generated parameters to be optimized, which is called the initial population, where each parameter pair is called an individual. Genetic algorithm encodes each individual in series to form chromosome, and determines the fitness function according to the optimization objective to calculate the fitness of each individual. Several individuals with high fitness values are selected from the initial population, and the chromosomes encoded by these individuals are crossed and mutated to form a new generation of individual populations. Then the fitness of each individual in the new population is calculated, and the above operations are performed repeatedly until the target value or the maximum number of iterations satisfying the fitness is met. In the iterative process, the genetic algorithm can preserve the individuals with good fitness values and eliminate the individuals with poor fitness. The new population not only inherits the information of the previous generation, but also is superior to the previous generation. Through continuous iteration, the parameters can be optimized.</p>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 Evaluation metics systems</title>
<p>In order to evaluate the predictive performance of the model, four evaluation indicators were used in this study, including RMSE, R2, MAE, MAPE, Mlogloss and Confusion matrix. The first four indexes are used to evaluate the prediction performance of TOC, and the latter two are used to evaluate the accuracy of rock facies identification.<disp-formula id="e15">
<mml:math id="m30">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>M</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:msqrt>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
<disp-formula id="e16">
<mml:math id="m31">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
<disp-formula id="e17">
<mml:math id="m32">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>P</mml:mi>
<mml:mi>E</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mfrac>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mfrac>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>Among them, <italic>y</italic>
<sub>
<italic>i</italic>
</sub> represents the true value of TOC, <inline-formula id="inf11">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted value of TOC, and <italic>n</italic> represents the number of TOC data. The lower the value of the above index represents the better performance of the prediction model.<disp-formula id="e18">
<mml:math id="m34">
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>where <italic>n</italic> represents the number of samples, <italic>i</italic> is the <italic>i</italic>th sample; <italic>m</italic> represents the number of classes, <italic>j</italic> is the <italic>j</italic>th category; <italic>y</italic>
<sub>
<italic>i,j</italic>
</sub> represents whether the <italic>i</italic>th sample belongs to the <italic>j</italic>th class, belongs to 1, else to 0; <italic>pi</italic>,<italic>j</italic> represents the probability that the prediction model predicts the <italic>i</italic>th sample as <italic>j</italic>.</p>
<p>The Confusion matrix is defined as:<disp-formula id="e19">
<mml:math id="m35">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="{" close="}" separators="|">
<mml:mrow>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mn>11</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mn>12</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mn>21</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mn>22</mml:mn>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22f1;</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mtable columnalign="center">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x2026;</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mtd>
<mml:mtd>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
</p>
<p>In the formula, <italic>m</italic> is the number of categories divided, the subscript represents the label, and <italic>n</italic>
<sub>
<italic>ij</italic>
</sub> represents the number of samples whose real label is <italic>i</italic> and predicted as <italic>j</italic>. The Confusion matrix can be used to obtain the prediction accuracy, the accuracy of each category (<italic>P</italic>), and the recall rate (<italic>R</italic>). The calculation formula is as follows:<disp-formula id="e20">
<mml:math id="m36">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>
<disp-formula id="e21">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>
<disp-formula id="e22">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo>&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(22)</label>
</disp-formula>
</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<sec id="s4-1">
<title>4.1 Lithofacies and characteristics</title>
<sec id="s4-1-1">
<title>4.1.1 Lithofacies</title>
<p>Based on core observation, according to the difference of sedimentary structure and structure, five lithofacies are developed in the shale of Yanchang Formation, which are argillaceous shale facies (AS), siltstone/very fine sandstone facies (SS), tuff facies (TUF), argillaceous shale facies with silty lamina (ASLS) and argillaceous shale facies with tuff lamina (ATLS). The above five rock facies are easy to identify at the core scale. AS are black, grayish black, fine particles (<xref ref-type="fig" rid="F3">Figure 3A</xref>), and do not develop or develop a small amount of silty or tuffaceous layers; SS is mainly gray and grayish white, and a very small amount of grayish black argillaceous bands are developed (<xref ref-type="fig" rid="F3">Figure 3B</xref>); TUF is grayish yellow, easily broken (<xref ref-type="fig" rid="F3">Figure 3C</xref>), relatively homogeneous, and basically does not develop other lithologic layers; ASLS are mainly gray-black argillaceous shale, with a large number of gray-white and gray silty layers distributed inside. The thickness of these layers is generally millimeter and centimeter (<xref ref-type="fig" rid="F3">Figure 3D</xref>), and the cumulative thickness of silty layer accounts for 20%&#x2013;50%. The main body of ATLS is black argillaceous shale, with a large number of yellow or grayish yellow tuffaceous laminae distributed inside. The laminae thickness is generally in the millimeter and centimeter levels (<xref ref-type="fig" rid="F3">Figure 3E</xref>). The cumulative thickness of the tuffaceous layer accounts for 20%&#x2013;50%. Based on the above principles, a columnar distribution map of rock facies in 12 wells was drawn.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The difference of sedimentary and structure on the core scale <bold>(A&#x2013;E)</bold>: The characteristics of different rock facies on the core <bold>(F)</bold> distribution and logging response of rock facies in coring section of well YY22.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g003.tif"/>
</fig>
</sec>
<sec id="s4-1-2">
<title>4.1.2 Distribution characteristics of TOC in lithofacies</title>
<p>In the TOC frequency distribution diagram of different lithofacies shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, there are differences in the distribution range of TOC in different lithofacies. The TOC of TUF and SS is low, mainly distributed below 2.0&#xa0;wt%, (<xref ref-type="fig" rid="F4">Figures 4A,B</xref> )and the TOC distribution of AS and ASLS is medium (<xref ref-type="fig" rid="F4">Figures 4C,D</xref>). The average values are 5.03&#xa0;wt% and 6.39&#xa0;wt% The above four lithofacies have no obvious exhibit skewed distribution with a long tail, and the data have good balance. In the Yanchang Formation shale, the TOC of the ATLS is generally high, and the numerical distribution range is from 3&#xa0;wt% to 22&#xa0;wt% (<xref ref-type="fig" rid="F4">Figure 4E</xref>). The frequency distribution guidance diagram of the lithofacies shows a weak skew distribution. Compared with <xref ref-type="fig" rid="F1">Figure 1C</xref>, the proportion of data greater than 8&#xa0;wt% is all increased, and the imbalance of data is weakened.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Frequency distribution histogram of TOC in TUFF <bold>(A)</bold>, SS <bold>(B)</bold>, AS <bold>(C)</bold>, ASLS <bold>(D)</bold> and ATLS <bold>(E)</bold>.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g004.tif"/>
</fig>
<p>The relative proportion of different rock facies in the shale section of the Yanchang Formation is the main reason for the unbalanced distribution of TOC data in <xref ref-type="fig" rid="F1">Figure 1C</xref>. In the shale interval, AS has the highest proportion of thickness, which can account for the total thickness of the shale interval. Secondly, the TOC distribution characteristics of ASLS samples are similar to those of AS, which causes the overall TOC data to be concentrated in the TOC intervals of the above two lithofacies, while in other distribution intervals, especially in the high-value TOC interval of ATLS, there are fewer samples, resulting in unbalanced distribution of data in <xref ref-type="fig" rid="F1">Figure 1C</xref>. The data imbalance of TOC sub-data set of rock facies obtained by classification is reduced, which is helpful for learning algorithm to obtain more accurate prediction model.</p>
</sec>
<sec id="s4-1-3">
<title>4.1.3 Relationship between TOC and logging response in different lithofacies</title>
<p>As mentioned above, the relationship between formation logging response and TOC is affected by other formation parameters, including mineral composition, elemental composition, and organic matter type. <xref ref-type="table" rid="T2">Table 2</xref> shows the differences in mineral composition and organic matter types between different rock facies.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Comparision of mineral Composition and Pyrolysis parameters for different lithofacies.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Lithofacies</th>
<th rowspan="2" align="center">Type</th>
<th colspan="4" align="center">Mineral composition</th>
<th colspan="3" align="center">Pyrolysis parameters</th>
</tr>
<tr>
<th align="center">Feldspar and quartz (%)</th>
<th align="center">Clay (%)</th>
<th align="center">Carbonate (%)</th>
<th align="center">Pyrite (%)</th>
<th align="center">S1 (mg/g)</th>
<th align="center">S2 (mg/g)</th>
<th align="center">S1/TOC &#xd7; 100 (mg/g TOC)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">TU</td>
<td align="center">Average</td>
<td align="center">82.26</td>
<td align="center">13.12</td>
<td align="center">3.76</td>
<td align="center">0.86</td>
<td align="center">0.32</td>
<td align="center">0.96</td>
<td align="center">26</td>
</tr>
<tr>
<td align="center">Range</td>
<td align="center">73.3&#x223c;87.8</td>
<td align="center">8&#x223c;17.5</td>
<td align="center">0&#x223c;9.2</td>
<td align="center">0&#x223c;2.5</td>
<td align="center">0.31&#x223c;0.32</td>
<td align="center">0.84&#x223c;1.07</td>
<td align="center">4&#x223c;48</td>
</tr>
<tr>
<td rowspan="2" align="center">ATSL</td>
<td align="center">Average</td>
<td align="center">41.57</td>
<td align="center">42.67</td>
<td align="center">6.18</td>
<td align="center">8.88</td>
<td align="center">3.75</td>
<td align="center">16.88</td>
<td align="center">62</td>
</tr>
<tr>
<td align="center">Range</td>
<td align="center">24.6&#x223c;63.1</td>
<td align="center">22&#x223c;56.5</td>
<td align="center">0&#x223c;17.3</td>
<td align="center">2.4&#x223c;24.9</td>
<td align="center">0.86&#x223c;6.4</td>
<td align="center">4.3&#x223c;48.29</td>
<td align="center">24&#x223c;108</td>
</tr>
<tr>
<td rowspan="2" align="center">AS</td>
<td align="center">Average</td>
<td align="center">33.74</td>
<td align="center">57.02</td>
<td align="center">6.78</td>
<td align="center">2.48</td>
<td align="center">3.65</td>
<td align="center">11.10</td>
<td align="center">70</td>
</tr>
<tr>
<td align="center">Range</td>
<td align="center">17.6&#x223c;48.5</td>
<td align="center">45.5&#x223c;75</td>
<td align="center">0&#x223c;18.3</td>
<td align="center">0.4&#x223c;7</td>
<td align="center">2.3&#x223c;5.54</td>
<td align="center">8.24&#x223c;16.32</td>
<td align="center">5&#x223c;109</td>
</tr>
<tr>
<td rowspan="2" align="center">ASLS</td>
<td align="center">Average</td>
<td align="center">45.66</td>
<td align="center">41.33</td>
<td align="center">11.22</td>
<td align="center">1.79</td>
<td align="center">4.75</td>
<td align="center">9.67</td>
<td align="center">116</td>
</tr>
<tr>
<td align="center">Range</td>
<td align="center">27&#x223c;63.8</td>
<td align="center">20.5&#x223c;56</td>
<td align="center">2.3&#x223c;36.3</td>
<td align="center">0.4&#x223c;6.3</td>
<td align="center">1.93&#x223c;6.14</td>
<td align="center">4.22&#x223c;15.2</td>
<td align="center">75&#x223c;302</td>
</tr>
<tr>
<td rowspan="2" align="center">SS</td>
<td align="center">Average</td>
<td align="center">57.17</td>
<td align="center">24.74</td>
<td align="center">17.22</td>
<td align="center">0.86</td>
<td align="center">2.15</td>
<td align="center">4.20</td>
<td align="center">173</td>
</tr>
<tr>
<td align="center">Range</td>
<td align="center">13.8&#x223c;82.4</td>
<td align="center">8.7&#x223c;37.5</td>
<td align="center">3.6&#x223c;50.5</td>
<td align="center">0&#x223c;4</td>
<td align="center">0.37&#x223c;6.2</td>
<td align="center">0.47&#x223c;15.71</td>
<td align="center">69&#x223c;398</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From the perspective of mineral composition, AS has the characteristics of high clay mineral content, low felsic content and medium pyrite content. TUF and SS are characterized by low clay mineral content, low pyrite and high felsic content. The difference is that TUF has high carbonate content and SS has high carbonate content. ASLS has obvious transitional characteristics between AS and SS, that is, clay mineral content, carbonate mineral content, clay mineral content and pyrite are all at a medium level; the main characteristic of ATSL is the highest content of pyrite, which can reach 8.9% on average, and other minerals are at a medium level.</p>
<p>Through pyrolysis data, it can be seen that S1 and S2 are higher in the three rock phases of ATSL, AS and ASLE, with an average value of more than 3.5&#xa0;mg/g and 9.5&#xa0;mg/g. The S1 and S2 values of SS are lower, with an average value of 2.15&#xa0;mg/g and 4.2&#xa0;mg/g. The S1 and S2 values of TUF are the lowest, and S1 and S2 are below 1&#xa0;mg/g. From the S1/TOC &#xd7; 100 index, the SS value is the highest, reaching an average of 173&#xa0;mg/g TOC, followed by ASLS (an average of 116&#xa0;mg/g TOC), The average value of AS and ATSL is about 65&#xa0;mg/g TOC, and TU is the lowest, only 26&#xa0;mg/g TOC. S1/TOC is often used to evaluate the oil content in rocks (<xref ref-type="bibr" rid="B19">Jarvie, 2008</xref>). Considering that shale oil may adsorb/dissolve in kerogen, the higher S1/TOC value is generally considered to be a higher content of movable oil, that is, the higher oil content in pores (<xref ref-type="bibr" rid="B20">Li et al., 2015</xref>).</p>
<p>In addition, <xref ref-type="bibr" rid="B32">Qiu et al. (2014)</xref> and <xref ref-type="bibr" rid="B1">Akhtar et al. (2018)</xref> studied the geochemical characteristics of tuff layers in the Yanchang Formation of the Ordos Basin and found that tuff layers generally have high U and Th contents. In a comparative study, <xref ref-type="bibr" rid="B24">Lu (2020)</xref> and <xref ref-type="bibr" rid="B45">Yin et al. (2017)</xref> found that the layers of siltstone or silty lamina in Zhangjiatan shale often have low U and Th content, while argillaceous shale has relatively high U and Th content.</p>
<p>
<xref ref-type="fig" rid="F5">Figure 5</xref> shows the logging response distribution of each rock facies in different TOC intervals, and the trend line is drawn by the connection of 50th perecentile point in each interval. Because tuff generally has the characteristics of hole enlargement (as shown in <xref ref-type="fig" rid="F3">Figure 3F</xref>, YY22 well 1310&#xa0;m), the logging response value has great uncertainty, so the relevant data of tuff are not drawn. It can be seen from <xref ref-type="fig" rid="F5">Figure 5</xref> that there are great differences in the logging response trend lines between different TOC intervals in different rock facies, which also shows that the differences in mineral composition, element composition and oil content of different rock facies will affect the relationship between logging response and TOC.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Distribution characteristics of GR <bold>(A)</bold>, DT <bold>(B)</bold>, DEN <bold>(C)</bold>, CNL <bold>(D)</bold>, logRt <bold>(E)</bold>, and U <bold>(F)</bold> in different TOC intervals of rock facies.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g005.tif"/>
</fig>
<p>From <xref ref-type="fig" rid="F5">Figure 5</xref>, the relationship between TOC and logging response in different lithofacies can be classified into two categories:</p>
<p>One is that the trend lines are similar in direction but not coincident, as shown in <xref ref-type="fig" rid="F5">Figures 5A&#x2013;F</xref>. In <xref ref-type="fig" rid="F5">Figures 5A, F</xref>, because ATLS has the highest U and Th content, it often has higher GR and U logging values under the same TOC conditions as other lithofacies. Similar SS and ASTL have lower U and Th than AS, which makes it have lower GR and U values. The difference in mineral composition may be the main reason for the inconsistency of the trend lines in <xref ref-type="fig" rid="F5">Figures 5B, C</xref>. For example, the high density of pyrite and carbonate makes ASTL and SS lithofacies have higher density values under the same TOC, and similar minerals also make ASTL and SS have lower acoustic time difference. The difference in oil content caused the non-coincidence of the trend line in <xref ref-type="fig" rid="F5">Figure 5D</xref>. Oil has a higher H&#x2b; content than kerogen, resulting in SS and ASLS with higher S1/TOC under the same TOC. Higher neutron porosity values, on the contrary, ASTL neutron porosity is low.</p>
<p>The second is the difference in the direction of the trend line, as shown in <xref ref-type="fig" rid="F5">Figure 5E</xref>, the resistivity logging response distribution in different TOC intervals. The obvious feature is that the trend line of ATLS lithofacies is not obvious, and even in some TOC intervals, the resistivity decreases with the increase of TOC. The high content of pyrite in ATLS may be a key factor in this phenomenon, which also causes the resistivity of ATLS to be generally lower than that of other rock phases under the same TOC. Secondly, the low content of clay minerals with good conductivity and high oil content also lead to higher resistivity of SS and ASLS than AS.</p>
<p>It can be seen that the introduction of rock facies in the analysis of TOC and logging response relationship can better understand the relationship between TOC and different logging responses, so that the relationship is less affected by shale formation factors such as mineral composition and element composition.</p>
</sec>
</sec>
<sec id="s4-2">
<title>4.2 TOC interpretation of LBCRM</title>
<sec id="s4-2-1">
<title>4.2.1 Model building</title>
<p>In this study, four SVR models of rock facies were constructed, which were AS, SS, ATLS and ASLS. For three purposes, the prediction model of tuff facies (TUF) was not constructed: 1) the phenomenon of borehole enlargement is obvious in this lithofacies, and the quality of logging data is poor; 2) The proportion of tuff facies in the Yanchang Formation reservoir is low, and the number of TOC test samples is small. 3) The TOC content of the lithofacies is generally low and the values are concentrated (<xref ref-type="fig" rid="F4">Figure 4A</xref>). Datas are derived from FY1, YY18, WY1, YY12, YY2, YY27, YY28, and B36 wells. The total number of data is 412, of which AS, SS, ATLS and ASLS are 214,55,71 and 72 respectively. The above data are randomly assigned to supervised training data sets and validation sets at a ratio of 1:4.</p>
<p>For the need of comparison, this study also constructed a prediction model under the uniform regression fitting mode. The same as the above data, the supervised data did not contain the relevant samples of tuff facies, and the supervised training data set and verification set were obtained from the corresponding data sets of the above four lithofacies. The supervised data include seven kinds of data such as <italic>AC</italic>, <italic>DEN</italic>, <italic>GR</italic>, <italic>Rt</italic>, <italic>PE</italic>, <italic>Th/K</italic>, <italic>U/Th</italic>, and TOC. The data normalization is carried out by the following formula:<disp-formula id="e23">
<mml:math id="m39">
<mml:mrow>
<mml:msubsup>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi mathvariant="italic">min</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(23)</label>
</disp-formula>
</p>
<p>Among them, the logging data uses the same maximum and minimum values in the above five supervised data sets. Since the TOC of the samples in SS is much lower than that of the other rock facies, in order to ensure the final prediction accuracy, the TOC of the SS supervised data set is normalized to [0,3], and the TOC of the remaining four supervised data sets is normalized to [0,30], which is normalized to [0,30] in the uniform regression fitting model.</p>
<p>In this study, SVR is used as the basic algorithm, and genetic algorithm is used to optimize the hyper-parameters in SVR. The optimized parameters include kernel function type and its key parameters (<xref ref-type="table" rid="T3">Table 3</xref>). The fitness function of the genetic algorithm is the cross-validation MSE of the training data (using the K-fold cross-validation method, K &#x3d; 3). The genetic algorithm uses the bidding model to select the optimal individual (the selection ratio is 0.2). After crossover and mutation operations, a new generation of population is formed. The number of populations in each generation is 150, and the number of iterations is 200. <xref ref-type="table" rid="T3">Table 3</xref> shows the range of hyperparameter optimization.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Optimal parameter series and cross validation root mean square error of SVM regressor obtained by genetic algorithm under different kernel function parameters.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="3" align="center">Kernel function type</th>
<th rowspan="3" align="center">Optimize parameters</th>
<th rowspan="3" align="center">Optimize range</th>
<th colspan="8" align="center">LBMRM</th>
<th rowspan="2" colspan="2" align="center">URM</th>
</tr>
<tr>
<th colspan="2" align="center">AS</th>
<th colspan="2" align="center">ATLS</th>
<th colspan="2" align="center">ASLS</th>
<th colspan="2" align="center">SS</th>
</tr>
<tr>
<th align="center">Best value</th>
<th align="center">CV MSE</th>
<th align="center">Best value</th>
<th align="center">CV MSE</th>
<th align="center">Best value</th>
<th align="center">CV MSE</th>
<th align="center">Best value</th>
<th align="center">CV MSE</th>
<th align="center">Best value</th>
<th align="center">CV MSE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="center">Polynomial</td>
<td align="center">&#x395;</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>&#x2212;5.71</sup>
</td>
<td rowspan="4" align="char" char=".">0.67</td>
<td align="center">2<sup>&#x2212;4.67</sup>
</td>
<td rowspan="4" align="char" char=".">2.28</td>
<td align="center">2<sup>&#x2212;5.48</sup>
</td>
<td rowspan="4" align="char" char=".">0.42</td>
<td align="center">2<sup>&#x2212;3.14</sup>
</td>
<td rowspan="4" align="char" char=".">0.18</td>
<td align="center">2<sup>&#x2212;6.29</sup>
</td>
<td rowspan="4" align="char" char=".">1.5</td>
</tr>
<tr>
<td align="center">&#x393;</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>&#x2212;2.80</sup>
</td>
<td align="center">2<sup>&#x2212;3.41</sup>
</td>
<td align="center">2<sup>&#x2212;2.88</sup>
</td>
<td align="center">2<sup>1.27</sup>
</td>
<td align="center">2<sup>&#x2212;3.48</sup>
</td>
</tr>
<tr>
<td align="center">C</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>4.87</sup>
</td>
<td align="center">2<sup>8.39</sup>
</td>
<td align="center">2<sup>5.66</sup>
</td>
<td align="center">2<sup>1.10</sup>
</td>
<td align="center">10</td>
</tr>
<tr>
<td align="center">D</td>
<td align="center">[2,10]</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
<td align="center">2</td>
</tr>
<tr>
<td rowspan="3" align="center">RBF</td>
<td align="center">&#x395;</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>&#x2212;5.79</sup>
</td>
<td rowspan="3" align="center">0.6</td>
<td align="center">2<sup>&#x2212;4.88</sup>
</td>
<td rowspan="3" align="center">2.49</td>
<td align="center">2<sup>&#x2212;7.92</sup>
</td>
<td rowspan="3" align="center">0.43</td>
<td align="center">2<sup>&#x2212;8.09</sup>
</td>
<td rowspan="3" align="center">0.11</td>
<td align="center">2<sup>&#x2212;5.35</sup>
</td>
<td rowspan="3" align="center">1.32</td>
</tr>
<tr>
<td align="center">&#x393;</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>&#x2212;5.21</sup>
</td>
<td align="center">2<sup>&#x2212;2.24</sup>
</td>
<td align="center">2<sup>&#x2212;1.33</sup>
</td>
<td align="center">2<sup>5.53</sup>
</td>
<td align="center">2<sup>2.92</sup>
</td>
</tr>
<tr>
<td align="center">C</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>7.08</sup>
</td>
<td align="center">2<sup>5.80</sup>
</td>
<td align="center">2<sup>5.38</sup>
</td>
<td align="center">2<sup>&#x2212;9.89</sup>
</td>
<td align="center">2<sup>&#x2212;1.04</sup>
</td>
</tr>
<tr>
<td rowspan="3" align="center">Sigmod</td>
<td align="center">&#x395;</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>&#x2212;5.94</sup>
</td>
<td rowspan="3" align="center">0.62</td>
<td align="center">2<sup>&#x2212;5.52</sup>
</td>
<td rowspan="3" align="center">3.01</td>
<td align="center">2<sup>&#x2212;5.36</sup>
</td>
<td rowspan="3" align="center">0.45</td>
<td align="center">2<sup>&#x2212;3.34</sup>
</td>
<td rowspan="3" align="center">0.148</td>
<td align="center">2<sup>&#x2212;6.03</sup>
</td>
<td rowspan="3" align="center">1.57</td>
</tr>
<tr>
<td align="center">&#x393;</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>&#x2212;6.64</sup>
</td>
<td align="center">2<sup>&#x2212;6.83</sup>
</td>
<td align="center">2<sup>&#x2212;7.58</sup>
</td>
<td align="center">2<sup>0.03</sup>
</td>
<td align="center">2<sup>&#x2212;5.83</sup>
</td>
</tr>
<tr>
<td align="center">C</td>
<td align="center">[2<sup>&#x2212;10</sup>,2<sup>10</sup>]</td>
<td align="center">2<sup>9.99</sup>
</td>
<td align="center">2<sup>9.99</sup>
</td>
<td align="center">2<sup>9.99</sup>
</td>
<td align="center">2<sup>2.43</sup>
</td>
<td align="center">2<sup>9.92</sup>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>SVR and genetic algorithm are implemented based on libsvm (<ext-link ext-link-type="uri" xlink:href="https://www.csie.ntu.edu.tw/%7Ecjlin/libsvm/">https://www.csie.ntu.edu.tw/&#x223c;cjlin/libsvm/</ext-link>) and geatpy package (<ext-link ext-link-type="uri" xlink:href="http://geatpy.com/index.php/quickstart/">http://geatpy.com/index.php/quickstart/</ext-link>).</p>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Performance of model prediction</title>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> shows the fitness curves of TOC interpretation model (<xref ref-type="fig" rid="F6">Figures 6A&#x2013;D</xref>) and homogeneous regression interpretation model (<xref ref-type="fig" rid="F6">Figure 6E</xref>) based on rock facies classification and regression under genetic algorithm optimization. <xref ref-type="table" rid="T4">Table 4</xref> lists the optimal parameters of the above optimization process and their corresponding cross validation MSE. It can be found from <xref ref-type="fig" rid="F6">Figure 6</xref> and <xref ref-type="table" rid="T4">Table 4</xref> that the RBF kernel function obtains higher cross-validation accuracy in both interpretation models, that is, the cross-validation MSE is the smallest, which is better than the other two kernel functions. The optimal cross MSE obtained by the RBF kernel function in the homogeneous regression interpretation model is 1.3. In the TOC interpretation model based on rock facies classification regression, the optimal cross MSE in the tuffaceous/clay interbedded shale data set is 2.18, and the optimal root mean square error of other data sets is less than 0.7.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Curve of the fitness with the genetic algorithm otimization in AS training datasets <bold>(A)</bold>, ATLS training datasets <bold>(B)</bold>, ASLS training datasets <bold>(C)</bold>, SS training datasets <bold>(D)</bold> and all data training datasets <bold>(E)</bold>.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g006.tif"/>
</fig>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Interpretation accuracy evaluation indexes of LBCRM and URM applied in different petrographic verification sets.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Evaluation metics</th>
<th colspan="5" align="center">LBCRM</th>
<th colspan="5" align="center">URM</th>
</tr>
<tr>
<th align="center">AS</th>
<th align="center">ATLS</th>
<th align="center">ASLS</th>
<th align="center">SS</th>
<th align="center">All testing data</th>
<th align="center">AS</th>
<th align="center">ATLS</th>
<th align="center">ASLS</th>
<th align="center">SS</th>
<th align="center">All testing data</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Sample Number</td>
<td align="center">43</td>
<td align="center">17</td>
<td align="center">15</td>
<td align="center">11</td>
<td align="center">86</td>
<td align="center">43</td>
<td align="center">17</td>
<td align="center">15</td>
<td align="center">11</td>
<td align="center">86</td>
</tr>
<tr>
<td align="left">MSE</td>
<td align="center">0.70</td>
<td align="center">2.24</td>
<td align="center">0.49</td>
<td align="center">0.20</td>
<td align="center">0.91</td>
<td align="center">1.15</td>
<td align="center">11.73</td>
<td align="center">0.59</td>
<td align="center">1.86</td>
<td align="center">3.23</td>
</tr>
<tr>
<td align="left">RMSE</td>
<td align="center">0.84</td>
<td align="center">1.49</td>
<td align="center">0.70</td>
<td align="center">0.37</td>
<td align="center">0.95</td>
<td align="center">1.07</td>
<td align="center">3.42</td>
<td align="center">0.77</td>
<td align="center">1.36</td>
<td align="center">1.80</td>
</tr>
<tr>
<td align="left">MAPE</td>
<td align="center">15.76</td>
<td align="center">16.37</td>
<td align="center">15.20</td>
<td align="center">37.84</td>
<td align="center">19.56</td>
<td align="center">16.32</td>
<td align="center">25.14</td>
<td align="center">15.90</td>
<td align="center">93.23</td>
<td align="center">31.65</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In order to better evaluate the generalization ability and prediction accuracy of the above model, the above model is applied in the corresponding validation set, and the TOC in each validation set data is predicted respectively. <xref ref-type="fig" rid="F7">Figure 7</xref> is the projection plot of the predicted TOC value and the measured TOC value of different prediction models. <xref ref-type="fig" rid="F7">Figure 7A</xref> is the projection plot of the prediction results of the uniform regression interpretation model in its response data set, and <xref ref-type="fig" rid="F7">Figure 7B</xref> is the projection plot of the prediction results of the classification regression interpretation model based on lithofacies in their respective validation data sets. <xref ref-type="fig" rid="F7">Figure 7A</xref> shows that the overall prediction effect of the uniform regression interpretation model is poor. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, MSE is 3.23, RMSE is 1.80, and MAPE is 31.65. It is worth noting that in the interval of TOC&#x3e;9% in <xref ref-type="fig" rid="F7">Figure 7A</xref>, the predicted value of the prediction model seriously deviates from the true value, and the relative error of the predicted value of individual data points is far more than 25%. In comparison, the classification regression interpretation model based on rock facies shown in <xref ref-type="fig" rid="F7">Figure 7B</xref> has better prediction performance. The distribution of data points is closer to the baseline of real TOC equal to predicted TOC. The MSE, RMSE and MAPE in the evaluation indexes are 0.91, 0.95 and 19.56, respectively. The prediction ability is greatly improved compared with the uniform regression model.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Comparison of measured and predicted TOC <bold>(A,C)</bold>: using uniform regression prediction model;<bold>(B,D)</bold>: classification regression prediction model based on rock facies. The different form of data points in Figures <bold>(C,D)</bold> represent lithofacies types. </p>
</caption>
<graphic xlink:href="feart-10-1106799-g007.tif"/>
</fig>
<p>Through the difference of prediction performance of the two interpretation models in different rock facies, the reason of poor prediction performance of homogeneous prediction model can be further analyzed. <xref ref-type="fig" rid="F7">Figure 7C</xref> shows the relationship between real and predicted TOC in different lithofacies under the homogeneous regression model. It can be seen from the figure that although the overall prediction performance is not good, the homogeneous regression model has higher prediction performance on AS and ASLS, and each evaluation index is at a lower level. The reason for the great decrease of prediction performance is ATLS and SS. The evaluation indexes MSE, RMSE and MAPE of ATLS prediction results are 11.73, 3.42 and 25.14 respectively. Considering the low TOC characteristics of SS, it may be more appropriate to use MAPE for evaluation, but MAPE &#x3d; 93.23 is obviously beyond the acceptable range. The reason for the above characteristics can be attributed to the fact that the uniform regression model makes the learning algorithm pay more attention to the learning of data features in the high distribution density interval. The SVR algorithm learns more feature information from the 3% &#x223c; 8% interval shown in <xref ref-type="fig" rid="F1">Figure 1C</xref> for training, so that the evaluation index MSE is minimized. In this case, the prediction model will extract features from the two lithofacies of AS and ASLS. <xref ref-type="fig" rid="F5">Figure 5</xref> shows that the characteristics of TOC and logging in different lithofacies have certain differences, which results in the decline of prediction performance in ATLS and SS.</p>
<p>The classification regression prediction model based on lithofacies solves the problem of poor prediction performance of single prediction model in ATLS and SS, making the prediction performance close to AS and ASLS. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, the prediction indexes of ATLS and SS have been greatly improved (<xref ref-type="fig" rid="F7">Figure 7D</xref>). The MSE, RMSE and MAPE of ASLS are 2.24.1.49 and 16.37 respectively, and SS is 0.20,0.37 and 37.84 respectively. At the same time, the prediction performance of AS and ASLE has also been improved. For example, the MSE, RMSE and MAPE of AS are reduced to 0.70,0.84 and 15.76, and ASLS is reduced to 0.49,0.70 and 15.20, respectively.</p>
<p>The above prediction results show that the classification regression interpretation based on lithofacies can obtain better TOC prediction accuracy in the case of data imbalance and multi-stratigraphic factors.</p>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 TOC computation in shale interval by LBCRM</title>
<sec id="s4-3-1">
<title>4.3.1 Model construction and performance evaluation of lithofacies test interpretation</title>
<p>The basis for the application of TOC interpretation model with better prediction performance is lithofacies interpretability. In this study, the lithofacies delineated in the coring section of 8 wells and their corresponding logging response values are used as supervised data sets, and the lithofacies logging recognition model is established by XGBoost algorithm.</p>
<p>The supervised data set includes seven kinds of logging data and lithofacies labels such as <italic>AC</italic>, <italic>DEN</italic>, <italic>GR</italic>, <italic>U</italic>, <italic>CAL</italic>, <italic>Th</italic> and <italic>Rt</italic>. The logging data uses the original data, and the lithofacies are coded according to TUFof 1, ATLS of 2, AS of 3, ASLS of 4, SS of 5. The XGboost algorithm is based on the XGboost package (<ext-link ext-link-type="uri" xlink:href="https://github.com/d%20mlc/xgboost">https://github.com/d mlc/xgboost</ext-link>). The base model type is gbtree, the learning task is multi: softmax, and the learning objective is mlogloss. In XGboost training, K-fold cross-validation is used to obtain the cross-validation mlogloss value to determine the optimal number of iterations of the tree in XGboost, where K &#x3d; 7, the maximum number is 200, and the optimal number of iterations is output after 30 iterations without performance improvement. Hyperparameters such as learning rate, max _ depth, min _ child _ weight, gamma and sub _ sample are optimized by genetic algorithm. The fitness function is set to cross mlogloss. The optimization range and coding method are shown in <xref ref-type="table" rid="T5">table 5</xref>.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Parameter to be optimized of XGboost and its optimal value in genetic algorithm.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">No.</th>
<th align="center">Parameter name</th>
<th align="center">Encoding type</th>
<th align="center">optimal range</th>
<th align="center">best value</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">1</td>
<td align="center">Learning rate</td>
<td align="center">Real number</td>
<td align="center">(0.3, 0.5]</td>
<td align="center">0.48</td>
</tr>
<tr>
<td align="center">2</td>
<td align="center">Max_depth</td>
<td align="center">Integer</td>
<td align="center">[5, 15]</td>
<td align="center">11</td>
</tr>
<tr>
<td align="center">3</td>
<td align="center">Min_child_weight</td>
<td align="center">Integer</td>
<td align="center">[5, 10]</td>
<td align="center">4</td>
</tr>
<tr>
<td align="center">4</td>
<td align="center">Gamma</td>
<td align="center">Real number</td>
<td align="center">[0, 0.4]</td>
<td align="center">0.1</td>
</tr>
<tr>
<td align="center">5</td>
<td align="center">Sub_sample</td>
<td align="center">Real number</td>
<td align="center">[0.7, 1]</td>
<td align="center">0.85</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="fig" rid="F8">Figure 8A</xref> shows the optimal and average fitness curves in the genetic algorithm optimization process, where the optimal mlogloss is 0.019, and the optimal parameters shown in <xref ref-type="table" rid="T5">Table 5</xref> are determined. The maximum number of times obtained by using this parameter is 179. Using the prediction model obtained by this parameter training, 556 data points in the coring sections of WY1 and FY3 were identified, and the confusion matrix shown in <xref ref-type="fig" rid="F8">Figure 8B</xref> was drawn (<xref ref-type="fig" rid="F8">Figure 8C</xref>).</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>
<bold>(A)</bold> The fitness change curve of genetic algorithm in XGBoost rock facies logging interpretation model optimization; <bold>(B)</bold> The confusion matrix of prediction results and measured results in WY1 and FY3 using XGboost rock facies logging interpretation model; <bold>(C)</bold> Comparison of rock facies prediction results and real results of Well WY1.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g008.tif"/>
</fig>
<p>The prediction accuracy of the model in the prediction set, the prediction accuracy and recall rate of each class are calculated by the confusion matrix (<xref ref-type="table" rid="T6">Table 6</xref>). It can be seen that the accuracy and recall rate of the four rock phases of TUF, ATLS, AS and SS are greater than 0.80, and the prediction accuracy and recall rate of ASLS are low, which can still reach about 0.75. The overall prediction accuracy is 0.86, which shows that the prediction model has high prediction accuracy and lays the foundation for the application of LBCRM model in actual drilling.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Evaluation metrics of XGboost classification model in validation set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">evaluation metics</th>
<th align="center">Tuff</th>
<th align="center">ATLS</th>
<th align="center">AS</th>
<th align="center">ASLS</th>
<th align="center">SS</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Recall</td>
<td align="center">0.81</td>
<td align="center">0.80</td>
<td align="center">0.90</td>
<td align="center">0.77</td>
<td align="center">0.88</td>
</tr>
<tr>
<td align="center">Precision</td>
<td align="center">0.81</td>
<td align="center">0.82</td>
<td align="center">0.89</td>
<td align="center">0.73</td>
<td align="center">0.91</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-3-2">
<title>4.3.2 Application for TOC content assessment</title>
<p>Based on lithofacies prediction model and LBCRM method, the TOC content of shale can be predicted. Firstly, the lithofacies prediction model is used to identify the lithofacies types of shale section, and the response TOC interpretation model is used to predict TOC for each lithofacies type. <xref ref-type="fig" rid="F9">Figure 9</xref> shows the lithofacies identification and TOC interpretation results of the shale depth section of 1,250 &#x223c; 1,340&#xa0;m in Well FY3. For comparison, the figure also shows the TOC calculated by a single regression model. It can be seen from <xref ref-type="fig" rid="F9">Figure 9</xref> that the TOC interpretation results obtained by the LBCRM, TOC interpretation model are closer to the measured values. Especially at 1,277 &#x223c; 1,283&#xa0;m, LBCRM successfully explained that in addition to the high TOC content in this depth section, the single regression model explained it as lower TOC.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>The prediction results of lithofacies and TOC in 1,250 &#x223c; 1,340&#xa0;m shale section of FY3.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g009.tif"/>
</fig>
</sec>
</sec>
<sec id="s4-4">
<title>4.4 Inspiration to model - Based interpretation</title>
<p>Model-based TOC interpretation is a method of TOC interpretation under appropriate assumptions and key parameters (<xref ref-type="bibr" rid="B18">Huang and Williamson, 1996</xref>; <xref ref-type="bibr" rid="B37">sondergeld et al., 2010</xref>). The strong heterogeneity of shale makes it difficult to accurately explain the TOC distribution of shale sections in the application of model-based interpretation methods. The LBCRM interpretation method based on the understanding of shale heterogeneity can better understand the model-based interpretation method and make the latter play a role in the logging interpretation of some key parameters.</p>
<p>&#x394;log<italic>R</italic> is the most classical and widely used model-based TOC interpretation method. <xref ref-type="bibr" rid="B31">Passey et al. (1990)</xref> pointed out that in immature shale, the resistivity curve is close to the base value, and &#x394;lg<italic>R</italic> is mainly provided by the amplitude of acoustic time difference deviating from the base value. In mature shale, both resistivity and acoustic travel time deviate from the baseline, and &#x394;lg<italic>R</italic> comes from the amplitude of the above two deviations from the base value. This also caused a statistically non-linear relationship between TOC, AC, Rt and maturity (Ro).</p>
<p>However, there are obvious differences in the content of conductive minerals and oil content between different rock phases, which will inevitably affect the relationship between &#x394;lg<italic>R</italic>-TOC by adding additional factors beyond maturity. For example, the relationship between resistivity and TOC shown in <xref ref-type="fig" rid="F5">Figure 5E</xref>, in the case of the same TOC, the ATLS has the characteristics of low resistivity and low acoustic time difference, and has the characteristics of low maturity in the &#x394;log<italic>R</italic> chart shown in <xref ref-type="fig" rid="F10">Figure 10A</xref> (<xref ref-type="bibr" rid="B31">Passey et al., 1990</xref>). The resistivity logging values of SS and ASLS are larger due to high oil content, which makes them have high maturity characteristics in the &#x394;log<italic>R</italic> chart, and the two lithofacies often cross multiple LOM intervals; in the &#x394;log<italic>R</italic> chart, the data points from AS are often concentrated in or near a certain LOM interval. The influence of shale heterogeneity on the &#x394;log method has prompted a large number of scholars to propose improved models (<xref ref-type="bibr" rid="B43">Wang et al., 2016</xref>; <xref ref-type="bibr" rid="B50">zhao et al., 2017</xref>), these methods without exception hope to expand the reference range of &#x394;log by selecting different baselines.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>
<bold>(A)</bold> The distribution characteristics of different rock facies on the &#x394;lg<italic>R</italic> chart; <bold>(B)</bold> Thermal evolution maturity prediction based on &#x394;log<italic>R</italic> plot using AS ata.</p>
</caption>
<graphic xlink:href="feart-10-1106799-g010.tif"/>
</fig>
<p>This paper does not focus on improving the traditional model-based prediction method to achieve better TOC prediction performance, but according to the characteristics of AS concentrated in a certain LOM interval in &#x394;log<italic>R</italic>, the combination of LBCRM and &#x394;log<italic>R</italic> is proposed to realize the logging estimation of Ro. Through the data distribution of AS on the &#x394;log<italic>R</italic>-TOC chart of YY22, YY1, W169 and DT5 wells (the data are all derived from the shale of Chang 7) (<xref ref-type="fig" rid="F10">Figure 10B</xref>), the LOM of the above 4 wells is estimated. Based on the conversion relationship between LOM and Ro by <xref ref-type="bibr" rid="B30">Passey (2010)</xref>, the Ro distribution interval can be estimated (<xref ref-type="table" rid="T7">Table 7</xref>), Compared with the measured Ro of four wells in this area by <xref ref-type="bibr" rid="B8">Cai et al. (2020)</xref>, the estimated value is close to the measured value, which fully shows that LBCRM can not only use logging to obtain more accurate TOC distribution in shale section, but also help geologists to explain more formation parameters after being used together with the model-based method.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Comparison of Ro based on AS data and &#x394;log<italic>R</italic> chart interpretation with real Ro.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Well name</th>
<th colspan="2" align="center">
<italic>&#x394;lgR interpretation</italic>
</th>
<th rowspan="2" align="center">Actal <italic>R</italic>
<sub>
<italic>O</italic>
</sub> (%) <xref ref-type="bibr" rid="B8">Cai et al., 2020</xref>)</th>
</tr>
<tr>
<th align="center">LOM</th>
<th align="center">
<italic>R</italic>
<sub>
<italic>O</italic>
</sub> (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">YY2</td>
<td align="center">10&#x223c;11</td>
<td align="center">0.82&#x223c;1.05</td>
<td align="center">1.10</td>
</tr>
<tr>
<td align="center">YY22</td>
<td align="center">10&#x223c;11</td>
<td align="center">0.82&#x223c;1.05</td>
<td align="center">1.05</td>
</tr>
<tr>
<td align="center">W169</td>
<td align="center">9&#x223c;10</td>
<td align="center">0.67&#x223c;0.82</td>
<td align="center">0.85</td>
</tr>
<tr>
<td align="center">DT5</td>
<td align="center">8&#x223c;9</td>
<td align="center">0.56&#x223c;0.67</td>
<td align="center">0.50</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>
<list list-type="simple">
<list-item>
<p>1) In this study, a TOC interpretation model based on lithofacies classification regression was proposed. Through the study of shale heterogeneity characteristics, this method can effectively reduce the influence of formation factors other than TOC on prediction accuracy by constructing TOC interpretation model for each rock facies category, and reduce the degree of data imbalance distribution, so that the data mining algorithm can achieve better prediction results.</p>
</list-item>
<list-item>
<p>2) The interpretability of lithofacies logging ensures the wellsite application based on the regression model of lithofacies classification. Compared with the traditional homogeneous regression model, the prediction performance is greatly improved, and the prediction of high TOC and low TOC sections is more accurate.</p>
</list-item>
<list-item>
<p>3) The LBCRM method based on the heterogeneity of shale can better understand the reasons for the deviation of traditional model-based interpretation methods. When combined with the latter, it can make the logging data provide more useful information.</p>
</list-item>
</list>
</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/Supplementary Material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>JY, CG and MC conceived and designed the experiments; QL, PX and SH performed the experiments; JY and CG analyzed the data; QZ contributed with figures l; JY, CG and MC wrote the paper. All authors read and approved the final manuscript.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This study was supported by the Major National Science and Technology Projects (No. 2017ZX05039001-005), the Research Project of YanchangOil Field Co., Ltd., 0 (No. ycsy2021jcts-B-06 and ycsy2022jcts-B-28), key R&#x26;D plan of Shaanxi Province (No. 2022GY-138,2021GY-113 and S2022-YF-YBGY-0471) and the National Natural Science Foundation of China (No. 41902136).</p>
<p>The authors declare that this study received funding from the Research Project of Yanchang Oil Field Co., Ltd.. The funder was not involved in the study design, collection, analysis, interpretation of data, the writing of this article or the decision to submit it for publication.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>Authors TJY, CG, SQL, PX, YSH and QZ were employed by Shaanxi Yanchang Petroleum (Group) Corp.Ltd.</p>
<p>The remaining author declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Akhtar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sahir</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X. Y.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Genesis of tuff interval and its uranium enrichment in upper triassic of ordos Basin, NW China</article-title>. <source>Acta Geochim.</source> <volume>37</volume>, <fpage>32</fpage>&#x2013;<lpage>46</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Aldrich</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Seidle</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Sweet spot&#x201d; identifcation and optimization in unconventional reservoirs</article-title>,&#x201d; in <source>AAPG datapages/search and discovery article &#x23;90323</source> (<publisher-loc>Salt Lake City, Utah</publisher-loc>.</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Alfred</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Vernik</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>A new petrophysical model for organic shales</article-title>,&#x201d; in <source>SPWLA 53rd annual logging symposium</source> (<publisher-loc>Colombia</publisher-loc>: <publisher-name>Cartagena</publisher-name>).</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altowairqi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Rezaee</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Urosevic</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Shale elastic property relationships as a function of total organic carbon content using synthetic samples</article-title>. <source>J. Pet. Sci. Eng.</source> <volume>133</volume>, <fpage>392</fpage>&#x2013;<lpage>400</lpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2015.06.028</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Branco</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Torgo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ribeiro</surname>
<given-names>R. P.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Rebagg: Resampled BAGGing for imbalanced regression</article-title>. <source>Proceeedings Mach. Learn. Res.</source> <volume>94</volume>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Branco</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Torgo</surname>
<given-names>Lu&#xed;s</given-names>
</name>
<name>
<surname>Ribeiro</surname>
<given-names>R. P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A survey of predictive modeling on imbalanced domains</article-title>. <source>ACM Comput. Surv. (CSUR)</source> <volume>49</volume> (<issue>2</issue>), <fpage>1</fpage>&#x2013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1145/2907070</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buda</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Maki</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mazurowski</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>A systematic study of the class imbalance problem in convolutional neural networks</article-title>. <source>Neural Netw.</source> <volume>106</volume>, <fpage>249</fpage>&#x2013;<lpage>259</lpage>. <pub-id pub-id-type="doi">10.1016/j.neunet.2018.07.011</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>Z. J.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>X. R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X. Z.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Characteristics and controlling factors of organic pores in the 7th member of Yanchang Formation shale in the Southeastern Ordos Basin (in Chinese)</article-title>. <source>Oil&#x26; Gas. Geol.</source> <volume>41</volume> (<issue>2</issue>), <fpage>367</fpage>&#x2013;<lpage>379</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carpentier</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Huc</surname>
<given-names>A. Y.</given-names>
</name>
<name>
<surname>Bessereau</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1991</year>). <article-title>Wireline logging and source rocks estimation of organic carbon by the Carbolog method</article-title>. <source>Log. Anal.</source> <volume>32</volume> (<issue>3</issue>), <fpage>279</fpage>&#x2013;<lpage>297</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chan</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Hassan</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Usman</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Humphrey</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Alzayer</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Duque</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Total organic carbon (TOC) quantification using artificial neural networks: Improved prediction by leveraging XRF data</article-title>. <source>J. Pet.sci.eng.</source> <volume>108</volume>, <fpage>109302</fpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2021.109302</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>T. Q.</given-names>
</name>
<name>
<surname>Guestrin</surname>
<given-names>C.</given-names>
</name>
</person-group> &#x201c;<article-title>XGBoost: A scalable tree boosting system</article-title>,&#x201d; in <conf-name>22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>, <conf-loc>New York, NY, USA</conf-loc>, <conf-date>2016</conf-date>, <fpage>785</fpage>&#x2013;<lpage>794</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Y. Y.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>D. C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>X. N.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Organic Geochemistry and Evaluation of the Shale of Yanchang Formation in Yanchang Exploration Area of Ordos Basin</article-title>. <source>Unconv. Oil Gas</source> <volume>7</volume> (<issue>1</issue>), <fpage>32</fpage>&#x2013;<lpage>37</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.2095-8471.2020.01.007</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Curtis</surname>
<given-names>J. B.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Fractured shale-gas systems</article-title>. <source>Am. Assoc. Pet. Geol. Bull.</source> <volume>86</volume>, <fpage>1921</fpage>&#x2013;<lpage>1938</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dellenbach</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Espitalie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lebreton</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>1983</year>). &#x201c;<article-title>Source rock logging</article-title>,&#x201d; in <source>Transactions of the SPWLA 8th European formation evaluation symposium</source> (<publisher-loc>London, UK</publisher-loc>).</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xiong</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Lithofacies and reservoir characteristics of the Lower Cretaceous continental Shahezi Shale in the Changling Fault Depression of Songliao Basin, NE China</article-title>. <source>Mar. Petroleum Geol.</source> <volume>98</volume>, <fpage>401</fpage>&#x2013;<lpage>421</lpage>. <pub-id pub-id-type="doi">10.1016/j.marpetgeo.2018.08.035</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>S. B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z. L.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Exploration prospect of shale gas with Permian transitional facies of some key areas in China</article-title>. <source>Pet. Geol. &#x26;Experiment</source> <volume>43</volume> (<issue>3</issue>), <fpage>377</fpage>&#x2013;<lpage>385</lpage>. <pub-id pub-id-type="doi">10.11781/sysydz202103377</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holland</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1973</year>). <article-title>Erratum: genetic algorithms and the optimal allocation of trials</article-title>. <source>SIAM J. Comput.</source> <volume>2</volume> (<issue>2</issue>), <fpage>88</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1137/0202009</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z. H.</given-names>
</name>
<name>
<surname>Williamson</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Artificial neural network modelling as an aid to source rock characterization</article-title>. <source>Mar. Petroleum Geol.</source> <volume>13</volume> (<issue>2</issue>), <fpage>277</fpage>&#x2013;<lpage>290</lpage>. <pub-id pub-id-type="doi">10.1016/0264-8172(95)00062-3</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Jarvie</surname>
<given-names>D. M.</given-names>
</name>
</person-group> (<year>2008</year>). <source>Unconventional shale resource plays: Shale-gas and shale-oil opportunities</source>. <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Fort Worth Business Press Meeting</publisher-name>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Y. L.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>X. T.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>S. S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Impact of hydrocarbon expulsion efficiency of continental shale upon shale oil accumulations in eastern China</article-title>. <source>Mar. Petroleum Geol.</source> <volume>59</volume>, <fpage>467</fpage>&#x2013;<lpage>479</lpage>. <pub-id pub-id-type="doi">10.1016/j.marpetgeo.2014.10.002</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>X. W.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Geological conditions and exploration potential for shale gas in Upper Permian Wujiaping Formation in the region of Western Hubei-eastern Chongqing[J]</article-title>. <source>Pet. Geol. &#x26;Experiment</source> <volume>43</volume> (<issue>3</issue>), <fpage>386</fpage>&#x2013;<lpage>394</lpage>. <pub-id pub-id-type="doi">10.11781/sysydz202103386</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>S. X.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Large-scale long-tailed recognition in an open world</source>. <publisher-loc>louisiana, LA, USA</publisher-loc>: <publisher-name>CVPR</publisher-name>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>H. C.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>S. P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>The research on the heterogeneity of shale formations and its controlling factors&#x2014;A case study of the second member of Funing Formation in Subei Basin[J]</article-title>. <source>Unconv. Oil Gas</source> <volume>9</volume> (<issue>04</issue>), <fpage>78</fpage>&#x2013;<lpage>90</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2020</year>). <comment>Master&#x27;s Thesis</comment>. <publisher-loc>Beijing, China</publisher-loc>: <publisher-name>China university of petroleum Beijing</publisher-name>, <fpage>141</fpage>.<article-title>Study on tuff reservoir characteristics and formation mechanism of the Chang 7 member in the southern Ordos Basin</article-title>,</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>Y. Z.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Unconventional resources from exploration to production</article-title>,&#x201d; in <source>Unconventional oil and gas resources handbook: Evaluation and development</source> (<publisher-loc>Netherlands, Europe</publisher-loc>: <publisher-name>Elsevier</publisher-name>), <fpage>3</fpage>&#x2013;<lpage>52</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mahmoud</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Elkatatny</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mahmoud</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abouelresh</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abdulraheem</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Determination of the total organic carbon (TOC) based on conventional well logs using artificial neural network</article-title>. <source>Int. J. Coal Geol.</source> <volume>179</volume>, <fpage>72</fpage>&#x2013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1016/j.coal.2017.05.012</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meng</surname>
<given-names>Q. Q.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J. Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W. H.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X. F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Simulation Study on the Effect of Gypsum-salt Content on Hydrocarbon Generation in Mature Stage Shale</article-title>. <source>Special Oil Gas Reservoirs</source> <volume>5</volume>, <fpage>113</fpage>&#x2013;<lpage>118</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meng</surname>
<given-names>Q. Q.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Identification method for the origin of natural hydrogen gas in geological bodies[J]</article-title>. <source>PETROLEUM Geol. Exp.</source> <volume>44</volume> (<issue>3</issue>), <fpage>552</fpage>&#x2013;<lpage>558</lpage>. <pub-id pub-id-type="doi">10.11781/sysydz202203552</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ou</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C. C.</given-names>
</name>
<name>
<surname>Rui</surname>
<given-names>Z. H.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Lithofacies distribution and gas-controlling characteristics of the Wufeng&#x2013;Longmaxi black shales in the southeastern region of the Sichuan Basin, China</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>165</volume>, <fpage>269</fpage>&#x2013;<lpage>283</lpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2018.02.024</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Passey</surname>
<given-names>Q. R.</given-names>
</name>
<name>
<surname>Bohacs</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Esch</surname>
<given-names>W. L.</given-names>
</name>
<name>
<surname>Klimentidis</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sinha</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>From oil-prone source rock to gas-producing shale reservoir-geologic and petrophysical characterization of unconventional shale gas reservoirs</article-title>,&#x201d; in <source>International oil and gas conference and exhibition in China</source> (<publisher-loc>London, UK</publisher-loc>: <publisher-name>Society of Petroleum Engineers</publisher-name>).</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Passey</surname>
<given-names>Q. R.</given-names>
</name>
<name>
<surname>Creaney</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kulla</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Moretti</surname>
<given-names>F. J.</given-names>
</name>
<name>
<surname>Stroud</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>A practical model for organic richness from porosity and resistivity logs</article-title>. <source>AAPG Bull.</source> <volume>74</volume> (<issue>12</issue>), <fpage>1777</fpage>&#x2013;<lpage>1794</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname>
<given-names>X. W.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C. Y.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>G. Z.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>F. F.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J. Q.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Late Triassic tuff intervals in the Ordos basin, Central China: Their depositional, petrographic, geochemical characteristics and regional implications</article-title>. <source>J. Asian Earth Sci.</source> <volume>80</volume>, <fpage>148</fpage>&#x2013;<lpage>160</lpage>. <pub-id pub-id-type="doi">10.1016/j.jseaes.2013.11.004</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schlanser</surname>
<given-names>K. M.</given-names>
</name>
</person-group> (<year>2015</year>). <source>Lithofacies classification in the Marcellus Shale and surrounding formations by applying Expectation Maximization to petrophysical and elastic well logs</source>. <publisher-loc>Laramie, WY, USA</publisher-loc>: <publisher-name>University of Wyoming</publisher-name>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmoker</surname>
<given-names>J. W.</given-names>
</name>
</person-group> (<year>1979</year>). <article-title>Determination of organic content of Appalachian Devonian shale from formation-density logs</article-title>. <source>AAPG Bull.</source> <volume>63</volume> (<issue>9</issue>), <fpage>1504</fpage>&#x2013;<lpage>1509</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schmoker</surname>
<given-names>J. W.</given-names>
</name>
</person-group> (<year>1981</year>). <article-title>Determination of organic matter content of Appalachian Devonian shale from gamma-ray logs</article-title>. <source>AAPG Bull.</source> <volume>65</volume> (<issue>7</issue>), <fpage>1285</fpage>&#x2013;<lpage>1298</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2008</year>). <source>Lithofacies and sequence stratigraphic framework of the barnett shale, northeastern Texas</source>. <comment>Ph.D. Dissertation</comment>. <publisher-loc>Norman, Oklahoma</publisher-loc>: <publisher-name>University of Oklahoma</publisher-name>, <fpage>81</fpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sondergeld</surname>
<given-names>C. H.</given-names>
</name>
<name>
<surname>Newsham</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Comisky</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Rice</surname>
<given-names>M. C.</given-names>
</name>
<name>
<surname>Rai</surname>
<given-names>C. S.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Petrophysical considerations in evaluating and producing shale gas resources</article-title>,&#x201d; in <source>SPE unconventional gas conference</source> (<publisher-loc>Pennsylvania, PA, USA</publisher-loc>: <publisher-name>Society of Petroleum Engineers</publisher-name>).</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Support-vector-regression machine technology for total organic carbon content prediction from wireline logs in organic shale: a comparative study</article-title>. <source>J. Nat. Gas. Sci. Eng.</source> <volume>26</volume> (<issue>1</issue>), <fpage>792</fpage>&#x2013;<lpage>802</lpage>. <pub-id pub-id-type="doi">10.1016/j.jngse.2015.07.008</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Vapnik</surname>
<given-names>V. N.</given-names>
</name>
</person-group> (<year>1995</year>). <source>The nature of statistical learning theory</source>. <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Springer-Verlag</publisher-name>, <fpage>188</fpage>.</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Carr</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Ju</surname>
<given-names>Y. W.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Statistical reverse model to predict mineral composition and TOC content of Marcellus shale</article-title>&#x201d;, in <conf-name>SPE Unconventional Resources Conference</conf-name>. <publisher-loc>Woodlands, TX, United States</publisher-loc>: <publisher-name>Society of petroleum Engineers</publisher-name>.</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>G. C.</given-names>
</name>
<name>
<surname>Carr</surname>
<given-names>T. R.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Methodology of organic-rich shale lithofacies identification and prediction: A case study from Marcellus Shale in the Appalachian basin</article-title>. <source>Comput. Geosciences</source> <volume>49</volume>, <fpage>151</fpage>&#x2013;<lpage>163</lpage>. <pub-id pub-id-type="doi">10.1016/j.cageo.2012.07.011</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An improved neural network for TOC, S1 and S2 estimation based on conventional well logs</article-title>. <source>J. Pet. Sci. Eng.</source> <volume>176</volume>, <fpage>664</fpage>&#x2013;<lpage>678</lpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2019.01.096</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Revised models for determining TOC in shale play: example from devonian Duvernay shale, Western Canada Sedimentary Basin</article-title>. <source>Mar. Pet. Geol.</source> <volume>70</volume>, <fpage>304</fpage>&#x2013;<lpage>319</lpage>. <pub-id pub-id-type="doi">10.1016/j.marpetgeo.2015.11.023</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X. B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>L. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Shale gas EUR estimation based on a probability method:a case study of infill wells in Jiaoshiba shale gas field[J]</article-title>. <source>Pet. Geol. &#x26;Experiment</source> <volume>43</volume> (<issue>1</issue>), <fpage>161</fpage>&#x2013;<lpage>168</lpage>. <pub-id pub-id-type="doi">10.11781/sysydz202101161</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Y. X.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>C. F.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Q. P.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Relationship between element geochemical characteristic and organic matter enrichment in Zhangjiatan Shale of Yanchang Formation,Ordos Basin</article-title>. <source>J. China Coal Soc.</source> <volume>42</volume> (<issue>6</issue>), <fpage>1544</fpage>&#x2013;<lpage>1556</lpage>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Rezaee</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Arif</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2017a</year>). <article-title>A new method for TOC estimation in tight shale gas reservoirs</article-title>. <source>Int. J. Coal Geol.</source> <volume>179</volume>, <fpage>269</fpage>&#x2013;<lpage>277</lpage>. <pub-id pub-id-type="doi">10.1016/j.coal.2017.06.011</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>Y. X.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>X. R.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X. Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L. X.</given-names>
</name>
<etal/>
</person-group> (<year>2017b</year>). <article-title>Study on the distribution of extractable organic matter in pores of lacustrine shale: an example of zhangjiatan shale from the upper triassic yanchang formation, ordos basin, China</article-title>. <source>Interpretation</source> <volume>5</volume> (<issue>2</issue>), <fpage>109</fpage>&#x2013;<lpage>126</lpage>. <pub-id pub-id-type="doi">10.1190/int-2016-0124.1</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y. Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>D. F.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>W. Q.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>W. W.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Systematic classification and characterization of small-scale sedimentary structure of the Wufeng Formation shale based on lithofacies&#x2014;&#x2014;Influence for the evaluation of deep shale reservoirs[J]</article-title>. <source>Unconv. Oil Gas</source> <volume>9</volume> (<issue>02</issue>), <fpage>26</fpage>&#x2013;<lpage>33</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>D. F.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y. H.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y. M.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>S. X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z. H.</given-names>
</name>
<name>
<surname>Jiao</surname>
<given-names>W. W.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Comments on the evaluation system of accurate evaluation and selection of deep marine shale reservoirs</article-title>. <source>Unconv. Oil Gas</source> <volume>9</volume> (<issue>02</issue>), <fpage>1</fpage>&#x2013;<lpage>7</lpage>.</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>P. Q.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>H. L.</given-names>
</name>
<name>
<surname>Rasouli</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W. H.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Z. H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>An improved model for estimating the TOC in shale formations</article-title>. <source>Mar. Pet. Geol.</source> <volume>83</volume>, <fpage>174</fpage>&#x2013;<lpage>183</lpage>. <pub-id pub-id-type="doi">10.1016/j.marpetgeo.2017.03.018</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhen</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>Caineng Zou</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Hongyan</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>Hongjie</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Shixin</given-names>
</name>
</person-group> (<year>2016</year>). <source>Lithofacies and organic geochemistry of the middle permian lucaogou Formation in the jimusar sag of the junggar basin</source>. <publisher-loc>NW China</publisher-loc>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname>
<given-names>D. Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S. X.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>M. C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Fully connected deep network: An improved method to predict TOC of shale reservoirs from well logs</article-title>. <source>Mar. Pet. Geol.</source> <volume>132</volume>, <fpage>105205</fpage>. <pub-id pub-id-type="doi">10.1016/j.marpetgeo.2021.105205</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>L. Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z. S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X. Q.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W. N.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>A new and reliable dual model- and data-driven TOC prediction concept: A TOC logging evaluation method using multiple overlapping methods integrated with semi-supervised deep learning</article-title>. <source>J. Pet.sci.eng.</source> <volume>188</volume>, <fpage>106944</fpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2020.106944</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>