<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Earth Sci.</journal-id>
<journal-title>Frontiers in Earth Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Earth Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-6463</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1540035</article-id>
<article-id pub-id-type="doi">10.3389/feart.2025.1540035</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Earth Science</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Study on lithology identification using a multi-objective optimization strategy to improve integrated learning models: a case study of the Permian Lucaogou Formation in the Jimusaer Depression</article-title>
<alt-title alt-title-type="left-running-head">Deng et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/feart.2025.1540035">10.3389/feart.2025.1540035</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Deng</surname>
<given-names>Xili</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Jiahong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Junkai</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2914326/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Feng</surname>
<given-names>Cheng</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2218953/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Research Institute of Petroleum Exploration and Development</institution>, <institution>PetroChina</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Faculty of Petroleum</institution>, <institution>China University of Petroleum-Beijing at Karamay</institution>, <addr-line>Karamay</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1902602/overview">Xin Sun</ext-link>, Sinopec Matrix Co., LTD, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2919042/overview">Meng Li</ext-link>, Xi&#x2019;an Shiyou University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2919616/overview">Zhongguo Yang</ext-link>, North China University of Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Cheng Feng, <email>fcvip0808@126.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>05</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1540035</elocation-id>
<history>
<date date-type="received">
<day>05</day>
<month>12</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Deng, Li, Chen and Feng.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Deng, Li, Chen and Feng</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Lithology identification is a critical task in logging interpretation and reservoir evaluation, with significant implications for recognizing oil and gas reservoirs. The challenge in shale reservoirs lies in the similar logging response characteristics of different lithologies and the imbalanced data scale, leading to fuzzy lithology classification boundaries and increased difficulty in identification. This study focuses on the shale reservoir of the Permian Lucaogou Formation in the Jimusaer Depression for lithology identification. Initially, a comprehensive sampling model&#x2014;Smote-Tomek (ST) is used to introduce new feature information into the dataset while removing redundant features, effectively addressing the issue of data imbalance. Then, by combining the multi-objective optimization strategy Artificial Rabbit Optimization (ARO) with the Light Gradient Boosting Machine (LightGBM) model, a new intelligent lithology identification model (ST-ARO-LightGBM) is proposed, aimed at solving the problem of non-optimal hyperparameter settings in the model. Finally, the proposed new intelligent lithology identification model is compared and analyzed with six models: K-Nearest Neighbors (KNN), Decision Tree (DT), Gradient Boosting Decision Tree (GBDT), Random Forest (RF), Extreme Gradient Boosting (XGBoost), and LightGBM, all after comprehensive sampling. The experimental results show that the ST-ARO-LightGBM model outperforms other classification models in terms of classification evaluation metrics for different lithologies, with an overall classification accuracy improvement of 9.13%. The method proposed in this paper can solve the problem of non-equilibrium in rock samples, and can further improve the classification performance of traditional machine learning, and provide a method reference for the lithology classification of shale reservoirs.</p>
</abstract>
<kwd-group>
<kwd>shale reservoir</kwd>
<kwd>lithology identification</kwd>
<kwd>multi-objective optimization</kwd>
<kwd>artificial rabbit optimization model</kwd>
<kwd>integrated learning model</kwd>
<kwd>comprehensive sampling</kwd>
</kwd-group>
<contract-num rid="cn001">No. 42364007 &#x7b2c;42364007&#x53f7;</contract-num>
<contract-num rid="cn002">&#x7f16;&#x53f7;:2021D01E22</contract-num>
<contract-sponsor id="cn001">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Natural Science Foundation of Xinjiang Uygur Autonomous Region<named-content content-type="fundref-id">10.13039/100009110</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Solid Earth Geophysics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Lithology identification is a critical task in the field of petroleum exploration and development, influencing reservoir evaluation and geological modeling processes. Logs data, characterized by high vertical resolution and continuity, is widely used for lithology identification (<xref ref-type="bibr" rid="B3">Baisakhi and Rima, 2018</xref>). Currently, tight and shale oil and gas have become significant alternative resources (<xref ref-type="bibr" rid="B13">Feng et al., 2020</xref>; <xref ref-type="bibr" rid="B14">Feng et al., 2021</xref>; <xref ref-type="bibr" rid="B12">Feng et al., 2023</xref>; <xref ref-type="bibr" rid="B39">Zou et al., 2015</xref>; <xref ref-type="bibr" rid="B23">Passey et al., 2010</xref>). However, due to the similar logging response characteristics of different lithologies within shale reservoirs and the ambiguity and subjectivity indicated by logs parameters, traditional lithology identification methods fail to effectively classify lithologies. Additionally, the limitations of coring data result in an imbalanced lithology dataset, further leading to inaccurate lithology prediction results. Machine learning methods can effectively alleviate these issues (<xref ref-type="bibr" rid="B2">Al-Anazi and Gates, 2010</xref>; <xref ref-type="bibr" rid="B29">Saporetti et al., 2019</xref>; <xref ref-type="bibr" rid="B4">Bestagini et al., 2017</xref>; <xref ref-type="bibr" rid="B6">Bressan et al., 2020a</xref>; <xref ref-type="bibr" rid="B18">Imamverdiyev and Lyudmila, 2019</xref>).</p>
<p>Machine learning, with its strong nonlinear mapping capabilities across multiple scales and dimensions, has been widely applied to the fine identification of lithologies (<xref ref-type="bibr" rid="B8">Chioma et al., 2018</xref>). Prabowo UN (<xref ref-type="bibr" rid="B25">Prabowo et al., 2023</xref>) used the KNN clustering algorithm to accurately classify different lithofacies types in thefield Z, Indonesia. And the influence of hyperparameter K in KNN model on lithology identification results is analyzed and compared. Li, Bressan, and others (<xref ref-type="bibr" rid="B20">Li et al., 2023</xref>; <xref ref-type="bibr" rid="B5">Bressan et al., 2020b</xref>) applied support vector machine models to lithology identification. Mou Dan (<xref ref-type="bibr" rid="B22">Mou et al., 2021</xref>) compared the accuracy and applicability of K-Nearest Neighbors, support vector machines, and adaptive boosting algorithms in identifying volcanic rock lithologies. As exploration efforts continue to increase, the lithology in actual reservoirs becomes more complex. The fitting effect of a single model is insufficient to accurately classify the lithology types of complex reservoirs. The emergence of integrated models, which combine the classification results of multiple single models, further improves the accuracy of lithology prediction. <xref ref-type="bibr" rid="B34">Thongsamea et al. (2021)</xref> and <xref ref-type="bibr" rid="B7">Chen et al. (2024)</xref> used conventional logging curves as inputs for the XGBoost model, accurately identifying the lithologies of volcanic reservoirs. <xref ref-type="bibr" rid="B17">Huang et al. (2023)</xref> and <xref ref-type="bibr" rid="B36">Wang et al. (2020)</xref> applied the Boosting algorithm integrated with the random forest model, effectively enhancing the accuracy of lithology identification. However, the accuracy of machine learning models in predicting lithology depends on the scale of the sample set (<xref ref-type="bibr" rid="B16">Han et al., 2024</xref>). Machine learning is insensitive to the feature parameters of minority class samples, and different combinations of hyperparameters can affect the model&#x2019;s lithology identification accuracy (<xref ref-type="bibr" rid="B28">Saporetti et al., 2021</xref>).</p>
<p>To address the above issues, this paper proposes a multi-objective optimization strategy to modify an integrated learning model (ST-ARO-LightGBM) for imbalanced sample datasets. This model incorporates ST comprehensive sampling technology to effectively solve the problem of sample imbalance. The ARO technology adaptively adjusts the model&#x2019;s hyperparameter combinations to find the optimal hyperparameter set, achieving efficient and accurate lithology identification of the shale reservoir in the Permian Lucaogou Formation of the Jimusaer Depression. By integrating advanced machine learning techniques and integrated sampling methods, our study demonstrates a general approach to enhanced lithology prediction that not only addresses the challenges of data imbalance and complex lithology identification in shale reservoirs, but also provides a robust solution that can be adapted to other regions with similar geological conditions.</p>
</sec>
<sec id="s2">
<title>2 Methods and theory</title>
<sec id="s2-1">
<title>2.1 Artificial Rabbit Optimization</title>
<p>ARO model is a novel intelligent multi-objective optimization strategy inspired by the group behaviors observed in the survival and evolution of rabbit populations (<xref ref-type="bibr" rid="B35">Wang et al., 2022</xref>). During their survival, rabbits primarily engage in two strategies: detouring foraging (exploration) and random digging (hiding). The transition between these strategies is mainly influenced by the rabbit&#x2019;s energy factor (<xref ref-type="disp-formula" rid="e1">Formula 1</xref>). When the energy factor is high, rabbits have more stamina and are more likely to adopt exploration strategies. Conversely, when the energy factor is low, they are more inclined to adopt hiding strategies to avoid predators.<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4</mml:mn>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>ln</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>In this context, A(<italic>t</italic>) represents the energy factor, which is a function that oscillates and gradually decreases over time <italic>t</italic>. <italic>T</italic> represents the total number of iterations of the algorithm, and <italic>r</italic> is a random number between 0 and 1.</p>
<p>The ARO algorithm simulates the survival and evolution process of a rabbit population to seek the optimal solution in a multi-dimensional space. Assume there is a rabbit population of N rabbits in a D-dimensional space, where the position of the <italic>i</italic>th rabbit can be represented as: <inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. During the iteration process of the population, the positions of the rabbits will continuously change. The mathematical model of the exploration strategy can be represented by <xref ref-type="disp-formula" rid="e2">Formula 2</xref>, and the mathematical model of the hiding strategy can be represented by <xref ref-type="disp-formula" rid="e3">Formula 3</xref>.<disp-formula id="e2">
<mml:math id="m3">
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mspace width="1em"/>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>&#x22c5;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>0.5</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m4">
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>sin</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mspace width="1em"/>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:mfrac>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mn>4</mml:mn>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>g</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the position of the <italic>i</italic>th rabbit at time <italic>t</italic>&#x2b;1, and <inline-formula id="inf3">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the position of the <italic>j</italic>th rabbit at time <italic>t</italic>; <italic>c</italic> and <italic>g</italic>
<sub>r</sub> represent sequences in the <italic>D</italic>-dimensional space where the <italic>m</italic>th position is 1 and all other positions are 0. <italic>n</italic>
<sub>1</sub> follows a standard normal distribution, and <italic>r</italic>
<sub>1</sub>,<italic>r</italic>
<sub>2</sub>,<italic>r</italic>
<sub>3</sub>,<italic>r</italic>
<sub>4</sub> are random numbers between 0 and 1.</p>
</sec>
<sec id="s2-2">
<title>2.2 Light Gradient Boosting Machine</title>
<p>LightGBM is a high-performance ensemble learning algorithm based on gradient boosting trees (<xref ref-type="bibr" rid="B19">Ke et al., 2017</xref>). Its core idea is to iteratively train multiple weak classifiers, where each iteration generates a new decision tree to correct the prediction errors of all previous trees, thereby gradually improving the overall predictive performance of the model. The objective function of LightGBM is as follows (<xref ref-type="disp-formula" rid="e4">Formula 4</xref>):<disp-formula id="e4">
<mml:math id="m7">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>b</mml:mi>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf4">
<mml:math id="m8">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mi>b</mml:mi>
<mml:msup>
<mml:mi>j</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the loss function at the <italic>t</italic>th iteration, <inline-formula id="inf5">
<mml:math id="m9">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the prediction error up to the (<italic>t</italic>-1)-th iteration, and <inline-formula id="inf6">
<mml:math id="m10">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mstyle>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the regularization term.</p>
<p>To improve training speed, LightGBM also employs the histogram-based optimization algorithm and the leaf-wise growth strategy. The histogram-based optimization algorithm reduces computational complexity by grouping continuous feature values into discrete bins, thereby decreasing the amount of computation required during the splitting of decision trees and more effectively determining the optimal split points. The leaf-wise growth strategy selects the leaf node with the maximum gain for splitting at each iteration. Compared to the traditional level-wise growth strategy, leaf-wise can quickly find good split nodes, effectively reducing the depth of the tree and speeding up model training. Additionally, the leaf-wise strategy can handle feature imbalance more flexibly, giving the model better generalization ability.</p>
</sec>
<sec id="s2-3">
<title>2.3 Smote-Tomek</title>
<p>During exploration, there is a significant imbalance in the number of lithological types in rock thin sections. Such severe sample imbalance can cause machine learning models to overly focus on the majority class samples and ignore the characteristics of the minority class samples during the learning process (<xref ref-type="bibr" rid="B9">Deng et al., 2023</xref>). To address this issue, sampling algorithms are needed to balance the number of samples of different lithologies, thereby enhancing the machine learning model&#x2019;s ability to analyze minority class samples. ST is a combined sampling algorithm that integrates Smote oversampling with Tomek link undersampling techniques (<xref ref-type="bibr" rid="B24">Pereira et al., 2020</xref>; <xref ref-type="bibr" rid="B11">Devi and Purkayastha, 2017</xref>). It has shown good effectiveness in addressing the problem of sample imbalance in datasets. Compared to single sampling methods, Smote-Tomek compensates for the limitations of the Smote oversampling method, which tends to focus only on the minority class samples, leading to further overlap of different types of samples and low-quality data synthesis. The specific implementation steps are as follows:<list list-type="simple">
<list-item>
<p>(1) For the minority class samples, calculate the distance between the minority class samples and other class samples.</p>
</list-item>
<list-item>
<p>(2) Based on the distance between samples, generate a new sample through linear interpolation, so that the new sample lies on the line connecting two samples</p>
</list-item>
<list-item>
<p>(3) In the newly generated dataset, recalculate the distances between different samples to find the nearest neighbor samples of different classes that form Tomek links, where the two samples in a Tomek link pair belong to different classes.</p>
</list-item>
<list-item>
<p>(4) Remove the majority class samples in the Tomek link pairs to reduce the overlap at the decision boundary of the dataset, decrease the number of majority class samples, and generate a high-quality dataset.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s2-4">
<title>2.4 ST-ARO-LightGBM model</title>
<p>The training effectiveness of a machine learning model primarily depends on the quality of the dataset and the configuration of model parameters (<xref ref-type="bibr" rid="B26">Probst et al., 2019</xref>). This paper first uses the ST model to perform combined sampling on the lithological dataset in the study area, generating a new dataset. Then, the ARO multi-objective optimization strategy is used to adjust the hyperparameters of the LightGBM model, enhancing its lithological identification capability. The steps to establish the ST-ARO-LightGBM model are as follows, with the flowchart shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.<list list-type="simple">
<list-item>
<p>1. Use the ST method on the collected imbalanced lithological dataset to generate a new dataset, making the number of different lithological samples relatively balanced.</p>
</list-item>
<list-item>
<p>2. Based on the number m of hyperparameters to be optimized in the LightGBM model and the range of these hyperparameters, initialize the positions of N artificial rabbit individuals in the ARO population <inline-formula id="inf7">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="[" close="]" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and initialize the number of algorithm iterations <italic>T.</italic>
</p>
</list-item>
<list-item>
<p>3. Split the combined sampled data into training and testing sets. Use the values of each artificial rabbit individual <italic>X</italic>
<sub>
<italic>i</italic>
</sub> as the input hyperparameters for the LightGBM model, establish the LightGBM lithology prediction model, and calculate the current model&#x2019;s fitness value <italic>F</italic>
<sub>
<italic>best</italic>
</sub> based on the testing set. The position <italic>X</italic>
<sub>
<italic>best</italic>
</sub> of the artificial rabbit individual corresponding to the highest fitness value is taken as the optimal hyperparameter combination for the LightGBM model.</p>
</list-item>
<list-item>
<p>4. For each artificial rabbit individual <italic>X</italic>
<sub>
<italic>i</italic>
</sub>, calculate the energy factor A. When A &#x3e; 1, the artificial rabbit adopts the exploration strategy shown in <xref ref-type="disp-formula" rid="e2">Formula 2</xref>. Similarly, when A &#x2264; 1, the artificial rabbit adopts the hiding strategy shown in <xref ref-type="disp-formula" rid="e3">Formula 3</xref>. This process updates the positions of the artificial rabbit population.</p>
</list-item>
<list-item>
<p>5. For each individual in the updated rabbit population, recalculate the fitness value of the corresponding LightGBM model. If the new highest fitness value <italic>F</italic>
<sub>
<italic>new</italic>
</sub> is greater than <italic>F</italic>
<sub>
<italic>best</italic>
</sub>, update the highest fitness value <italic>F</italic>
<sub>
<italic>best</italic>
</sub> and the optimal hyperparameter combination <italic>X</italic>
<sub>
<italic>best</italic>
</sub>; otherwise, take no action.</p>
</list-item>
<list-item>
<p>6. Repeat steps (3) to (5) until the maximum number of iterations of the algorithm is reached. Use the optimal hyperparameter combination <italic>X</italic>
<sub>
<italic>best</italic>
</sub> as the input for the LightGBM model to obtain the optimal ST-ARO-LightGBM lithology identification model.</p>
</list-item>
</list>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Flowchart of ST-ARO-LightGBM model establishment.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g001.tif"/>
</fig>
</sec>
<sec id="s2-5">
<title>2.5 Model evaluation metrics</title>
<p>In classification tasks, the confusion matrix is commonly used to reflect the relationship between true classes and predicted classes (<xref ref-type="bibr" rid="B32">Sun Y. et al., 2020</xref>). For example, in binary classification tasks as shown in <xref ref-type="table" rid="T1">Table 1</xref>, the number of samples where the true class is predicted as the true class is defined as True Positive (TP). Similarly, False Negative (FN), False Positive (FP), and True Negative (TN) are defined accordingly. Based on this, four model evaluation metrics can be defined, as shown in <xref ref-type="disp-formula" rid="e5">Formula 5</xref>: Accuracy, Recall, Precision, and F1-score. Accuracy reflects the overall performance of the model, while F1-score is the harmonic mean of Recall and Precision. Higher values of these evaluation metrics indicate better classification performance of the model.<disp-formula id="e5">
<mml:math id="m12">
<mml:mrow>
<mml:mfenced open="{" close="" separators="|">
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>Accuracy</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>TN</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>TN</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mtext>TP</mml:mtext>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FP</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mtext>Recall</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mtext>TP</mml:mtext>
<mml:mrow>
<mml:mtext>TP</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>FN</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi mathvariant="normal">F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>score</mml:mtext>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>Precision</mml:mtext>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>Recall</mml:mtext>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Confusion matrix for binary classification tasks.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Actual category</th>
<th colspan="2" align="center">Predicted category</th>
</tr>
<tr>
<th align="center">Positive category</th>
<th align="center">Negative category</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Positive category</td>
<td align="center">True Positive (TP)</td>
<td align="center">False Negative (FN)</td>
</tr>
<tr>
<td align="center">Negative category</td>
<td align="center">False Positive (FP)</td>
<td align="center">True Negative (TN)</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3">
<title>3 Experimental dataset</title>
<sec id="s3-1">
<title>3.1 Lithology types</title>
<p>The data of rock slices used in this paper are from Jimusaer Depression, as shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. The Jimusaer Depression is located in the eastern part of the Junggar Basin and overall presents a west-low east-high, west-faulted east-overlapping graben feature Lucaogou Formation in Jimusaer Depression is rich in tight oil and shale oil resources, and its shale formations are developed in two oil-bearing systems, the upper and the lower, which are characterized by the integration of source and reservoir, thin layer superposition, large thickness, whole oil-bearing and continuous distribution. Influenced by multi-source mixing and frequent changes in water bodies, the lithology of the Permian Lucaogou Formation in the study area exhibits complex and diverse characteristics (<xref ref-type="bibr" rid="B38">Zha, 2022</xref>; <xref ref-type="bibr" rid="B37">Xiong et al., 2023</xref>). As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. Based on thin section and core sample data, the lithology of the Permian Lucaogou Formation is divided into four categories according to grain size and mineral composition: mudstone, cloud-bearing sandstone, siltstone, and detrital-bearing dolomite. These categories are the subjects of this study, with sample proportions of 39.49%, 38.03%, 16.00%, and 6.49%, respectively.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Geological structure map of the Jimusaer Depression.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g002.tif"/>
</fig>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Pie chart of core lithology in the Permian Lucaogou Formation.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g003.tif"/>
</fig>
<p>The ratio of the majority class, mudstone, to the minority class, detrital-bearing dolomite, is close to 7:1, indicating a serious imbalance in the dataset. This imbalance can cause the model to overlook the feature extraction of the detrital-bearing dolomite class during training, thereby affecting the model&#x2019;s performance. This issue will be addressed in <xref ref-type="sec" rid="s4-1">Section 4.1</xref>.</p>
</sec>
<sec id="s3-2">
<title>3.2 Lithological logging response characteristics</title>
<p>The logging response characteristics of different rock types exhibit certain differences. Conventional logging curves are a comprehensive response to the mineral composition of rocks, the nature of pore fluids, and physical properties, while nuclear magnetic resonance (NMR) logging data can reflect physical factors such as the specific surface area and shape of rock pores (<xref ref-type="bibr" rid="B31">Singh and Maheswar, 2022</xref>; <xref ref-type="bibr" rid="B21">Mitchell, 2020</xref>). Therefore, combining conventional logging curves with NMR logging data after thin section correlation, eight curves were selected to establish the lithological dataset: Acoustic travel time (AC), Compensated Neutron Log (CNL), Density log (DEN), Natural Gamma Ray (GR), Deep Resistivity (RT), Shallow Resistivity (RXO), T2 geometric mean (T<sub>2LM</sub>), and Total Porosity from NMR (POR).</p>
<p>The study of the distribution of logging response parameters for different lithologies is fundamental for lithological identification. Therefore, it is necessary to statistically analyze the logging parameters of different lithologies within the study area, as shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Statistics on the range of logging response parameters for different lithologies.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">AC (us/ft)</th>
<th align="center">CNL (%)</th>
<th align="center">DEN (g/cm3)</th>
<th align="center">GR (gAPI)</th>
<th align="center">Log (RT)</th>
<th align="center">Log (RXO)</th>
<th align="center">Log (T<sub>2LM</sub>)</th>
<th align="center">POR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="2" align="center">mudstone</td>
<td align="center">58.13&#x2013;98.01</td>
<td align="center">11.38&#x2013;36.64</td>
<td align="center">2.30&#x2013;2.75</td>
<td align="center">37.05&#x2013;153.50</td>
<td align="center">&#x2212;0.78&#x2013;3.22</td>
<td align="center">&#x2212;0.94&#x2013;3.30</td>
<td align="center">&#x2212;0.02&#x2013;1.74</td>
<td align="center">0.005&#x2013;0.201</td>
</tr>
<tr>
<td align="center">(73.46)</td>
<td align="center">(23.79)</td>
<td align="center">(2.46)</td>
<td align="center">(80.88)</td>
<td align="center">(1.48)</td>
<td align="center">(1.35)</td>
<td align="center">(0.84)</td>
<td align="center">(0.081)</td>
</tr>
<tr>
<td rowspan="2" align="center">cloud-bearing sandstone</td>
<td align="center">60.36&#x2013;106.31</td>
<td align="center">14.23&#x2013;40.52</td>
<td align="center">2.22&#x2013;2.55</td>
<td align="center">35.61&#x2013;162.50</td>
<td align="center">0.72&#x2013;3.27</td>
<td align="center">0.30&#x2013;3.29</td>
<td align="center">&#x2212;0.04&#x2013;1.76</td>
<td align="center">0.021&#x2013;0.210</td>
</tr>
<tr>
<td align="center">(74.79)</td>
<td align="center">(24.97)</td>
<td align="center">(2.44)</td>
<td align="center">(76.21)</td>
<td align="center">(1.69)</td>
<td align="center">(1.50)</td>
<td align="center">(0.95)</td>
<td align="center">(0.102)</td>
</tr>
<tr>
<td rowspan="2" align="center">siltstone</td>
<td align="center">62.65&#x2013;101.41</td>
<td align="center">13.60&#x2013;41.98</td>
<td align="center">2.23&#x2013;2.58</td>
<td align="center">48.33&#x2013;132.94</td>
<td align="center">0.06&#x2013;3.00</td>
<td align="center">&#x2212;0.32&#x2013;2.71</td>
<td align="center">0.12&#x2013;1.96</td>
<td align="center">0.003&#x2013;0.179</td>
</tr>
<tr>
<td align="center">(72.30)</td>
<td align="center">(23.31)</td>
<td align="center">(2.44)</td>
<td align="center">(81.59)</td>
<td align="center">(1.42)</td>
<td align="center">(1.03)</td>
<td align="center">(1.18)</td>
<td align="center">(0.095)</td>
</tr>
<tr>
<td rowspan="2" align="center">detrital-bearing dolomite</td>
<td align="center">61.61&#x2013;111.15</td>
<td align="center">9.58&#x2013;43.54</td>
<td align="center">2.07&#x2013;2.67</td>
<td align="center">30.16&#x2013;120.44</td>
<td align="center">0.88&#x2013;2.95</td>
<td align="center">0.58&#x2013;3.11</td>
<td align="center">0.16&#x2013;1.77</td>
<td align="center">0.032&#x2013;0.190</td>
</tr>
<tr>
<td align="center">(76.51)</td>
<td align="center">(25.88)</td>
<td align="center">(2.44)</td>
<td align="center">(72.55)</td>
<td align="center">(1.76)</td>
<td align="center">(1.50)</td>
<td align="center">(0.79)</td>
<td align="center">(0.076)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Note: RT, RXO, and T<sub>2LM</sub> are all log-transformed, with the data format being <inline-formula id="inf8">
<mml:math id="m13">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mo>&#x2013;</mml:mo>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mtext>average</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, the box diagram can better describe the distribution range of logging parameters for each lithology. The lower boundary of the box represents the first quartile, i.e. 25% of the data is less than or equal to this value, the upper boundary represents the third quartile, and the black line in the middle of the box represents the median. The box plots of logging parameters for different lithologies indicate that mudstone exhibits characteristics of medium-high DEN, medium-high GR, medium-low RT, and medium-low POR. Cloud-bearing sandstone shows characteristics of medium AC, medium-low GR, medium-high RXO, and medium-high POR. Siltstone displays low AC, medium-high GR, medium-low RXO, and medium-high T<sub>2LM</sub>. Detrital-bearing dolomite exhibits medium-low GR, medium-high RT, medium-low T<sub>2LM</sub>, and medium-low POR. Different logging parameters reflect different physical information of the rocks (<xref ref-type="bibr" rid="B30">Sebtosheikh et al., 2015</xref>). However, there is no distinct separation between different lithologies based on single logging parameters alone. Therefore, it is necessary to integrate multiple logging parameters and use multidimensional, multi-scale machine learning methods for efficient data analysis to classify lithologies.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Box plots of logging parameters for different lithologies.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g004.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments</title>
<sec id="s4-1">
<title>4.1 Comprehensive sampling effect</title>
<p>In this study, a total of 895 core thin section samples were collected, including 353 mudstone samples, 341 cloud-bearing sandstone samples, 143 siltstone samples, and 58 detrital-bearing dolomite samples. The imbalance in sample numbers can cause the classifier to fail to adequately extract features from minority class samples, thus affecting its ability to classify minority class samples and reducing the model&#x2019;s generalizability and accuracy. To address this issue, the ST algorithm was employed for comprehensive sampling of the dataset. The sampling results are shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. The number of samples for siltstone and detrital-bearing dolomite increased, introducing new feature information to the dataset. The number of samples for mudstone and cloud-bearing sandstone slightly decreased, removing some redundant features of the majority classes. The overall dataset has now reached a balanced state after sampling.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Comparison of the number of samples before and after combined sampling.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g005.tif"/>
</fig>
<p>To verify the impact of ST combined sampling on the performance of the LightGBM model, this study splits the datasets before and after ST combined sampling into training and testing sets in a 7:3 ratio and trains the model using five-fold cross validation. <xref ref-type="table" rid="T3">Table 3</xref> shows the comparison of various evaluation metrics of the LightGBM model before and after ST combined sampling. Compared to before ST sampling, the model accuracy improved by 14.17%. The F1 scores for cloud-bearing sandstone, siltstone, and detrital-bearing dolomite increased by 1.31%, 6.92%, and 82.34%, respectively, while the F1 score for mudstone decreased by 7.22%. This decrease is because, before ST sampling, the model overly focused on mudstone samples, lacking sufficient learning of the characteristics of detrital-bearing dolomite. Consequently, the F1 score for mudstone slightly decreased, while the F1 score for detrital-bearing dolomite significantly increased.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Impact of ST composite sampling on classification performance of the LightGBM model.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">LightGBM</th>
<th align="center">Rock type</th>
<th align="center">Precision (%)</th>
<th align="center">Recall (%)</th>
<th align="center">F1 Score (%)</th>
<th align="center">Accuracy (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="center">Before ST Sampling</td>
<td align="center">mudstone</td>
<td align="center">68.91</td>
<td align="center">78.85</td>
<td align="center">73.54</td>
<td rowspan="4" align="center">60.67</td>
</tr>
<tr>
<td align="center">cloud-bearing sandstone</td>
<td align="center">69.64</td>
<td align="center">82.11</td>
<td align="center">75.36</td>
</tr>
<tr>
<td align="center">siltstone</td>
<td align="center">91.18</td>
<td align="center">70.45</td>
<td align="center">79.49</td>
</tr>
<tr>
<td align="center">detrital-bearing dolomite</td>
<td align="center">75.00</td>
<td align="center">11.54</td>
<td align="center">0.20</td>
</tr>
<tr>
<td rowspan="4" align="center">After ST Sampling</td>
<td align="center">mudstone</td>
<td align="center">68.09</td>
<td align="center">64.65</td>
<td align="center">66.32</td>
<td rowspan="4" align="center">74.84</td>
</tr>
<tr>
<td align="center">cloud-bearing sandstone</td>
<td align="center">81.18</td>
<td align="center">72.63</td>
<td align="center">76.67</td>
</tr>
<tr>
<td align="center">siltstone</td>
<td align="center">85.58</td>
<td align="center">87.25</td>
<td align="center">86.41</td>
</tr>
<tr>
<td align="center">detrital-bearing dolomite</td>
<td align="center">77.23</td>
<td align="center">88.64</td>
<td align="center">82.54</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2">
<title>4.2 ST-ARO-LightGBM model performance</title>
<p>The training effectiveness of the LightGBM model depends on the influence of multiple input hyperparameters. Therefore, the ARO algorithm was employed to find the optimal hyperparameter combination for the LightGBM model. The main hyperparameters and their default values are listed in <xref ref-type="table" rid="T4">Table 4</xref>. This study simultaneously optimized 7 hyperparameters of the model, where the first three parameters affect the ensemble process of the LightGBM model, and the remaining four parameters influence the process of generating weak classifiers. The ARO algorithm was set with a population size of 50, with each &#x201c;artificial rabbit&#x201d; containing 7 hyperparameters. The F1 score of the LightGBM model was used as the fitness value, and the maximum number of iterations was set to 50.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Main hyperparameters of the LightGBM model.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Hyperparameterization</th>
<th align="center">Default value</th>
<th align="center">Parameter meaning</th>
<th align="center">Search interval</th>
<th align="center">Global optimum</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">n_estimators</td>
<td align="center">10</td>
<td align="center">Number of base learners</td>
<td align="center">[10&#x2013;300]</td>
<td align="center">97</td>
</tr>
<tr>
<td align="center">learning_rate</td>
<td align="center">0.1</td>
<td align="center">learning rate</td>
<td align="center">[0.001&#x2013;1]</td>
<td align="center">0.3</td>
</tr>
<tr>
<td align="center">subsample</td>
<td align="center">1.0</td>
<td align="center">Proportion of training samples</td>
<td align="center">[0.1&#x2013;1]</td>
<td align="center">0.61</td>
</tr>
<tr>
<td align="center">max_depth</td>
<td align="center">&#x2212;1</td>
<td align="center">Maximum depth of the tree</td>
<td align="center">[1&#x2013;30]</td>
<td align="center">17</td>
</tr>
<tr>
<td align="center">min_child_weight</td>
<td align="center">0.001</td>
<td align="center">The minimum weight required for cotyledon nodes</td>
<td align="center">[0.001&#x2013;1]</td>
<td align="center">0.03</td>
</tr>
<tr>
<td align="center">min_child_samples</td>
<td align="center">20</td>
<td align="center">The minimum number of samples required for cotyledon nodes</td>
<td align="center">[10&#x2013;50]</td>
<td align="center">10</td>
</tr>
<tr>
<td align="center">num_leaves</td>
<td align="center">31</td>
<td align="center">The maximum number of leaf nodes for the base learner</td>
<td align="center">[3&#x2013;100]</td>
<td align="center">89</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="fig" rid="F6">Figure 6</xref> shows the change in F1 score during the iteration process with the blue line, while the yellow baseline represents the F1 score trained with default parameters. After 21 iterations, the F1 score stabilized at 81.06%, which is a 3.08% improvement compared to the baseline. After the iteration, the optimal hyperparameter combination of the model is found, and the global optimal solution is shown in the last column of <xref ref-type="table" rid="T4">Table 4</xref>. At this point, the optimal model contains 97 decision trees, the maximum depth of the trees is 17, the minimum weight to control the splitting of leaf nodes is 0.03, the minimum sample number is 10, and the maximum number of leaf nodes in each decision tree is 89. The proportion of training samples is 0.61, and the learning rate of the model is 0.3.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The change of F1 value of ST-ARO-LightGBM identification lithology with the number of iterations.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g006.tif"/>
</fig>
<p>After dividing the dataset following ST sampling into training and testing sets in a 7:3 ratio, the ST-ARO-LightGBM lithology recognition model was established using AC, CNL, DEN, GR, RT, RXO, T<sub>2LM</sub>, POR eight curves, and the optimal hyperparameter combination as inputs. <xref ref-type="table" rid="T5">Table 5</xref> shows the classification performance of the proposed model on different lithologies. The classification accuracy on the test set is 81.25%, with F1 scores for the four lithologies being 70.21%, 79.35%, 87.62%, and 87.10% respectively, averaging 81.07%. Compared to the untuned ST-LightGBM model (<xref ref-type="table" rid="T3">Table 3</xref>), precision, recall, F1 score, and accuracy have improved by 2.99%, 3.15%, 3.09%, and 6.41%, respectively, demonstrating that the ARO algorithm effectively enhances various evaluation metrics of the LightGBM model.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Performance of the ST-ARO-LightGBM model for the identification of different lithologies.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">Precision (%)</th>
<th align="center">Recall (%)</th>
<th align="center">F1 Score (%)</th>
<th align="center">Accuracy (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">mudstone</td>
<td align="center">74.16</td>
<td align="center">66.67</td>
<td align="center">70.21</td>
<td rowspan="5" align="center">81.25</td>
</tr>
<tr>
<td align="left">cloud-bearing sandstone</td>
<td align="center">82.02</td>
<td align="center">76.84</td>
<td align="center">79.35</td>
</tr>
<tr>
<td align="left">siltstone</td>
<td align="center">85.19</td>
<td align="center">90.20</td>
<td align="center">87.62</td>
</tr>
<tr>
<td align="left">Detrital-bearing dolomite</td>
<td align="center">82.65</td>
<td align="center">92.05</td>
<td align="center">87.10</td>
</tr>
<tr>
<td align="left">average value</td>
<td align="center">81.01</td>
<td align="center">81.44</td>
<td align="center">81.07</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<sec id="s5-1">
<title>5.1 Model comparison</title>
<p>To verify the lithological classification capabilities of different machine learning models for the Permian shale reservoirs in Jimusar. KNN, DT, XGBoost, GBDT, and RF are several classical machine learning models that have been applied in lithology identification by predecessors (<xref ref-type="bibr" rid="B15">Guo et al., 2003</xref>; <xref ref-type="bibr" rid="B27">Ren et al., 2023</xref>; <xref ref-type="bibr" rid="B10">Dev and Eden, 2019</xref>; <xref ref-type="bibr" rid="B33">Sun Z. et al., 2020</xref>; <xref ref-type="bibr" rid="B1">Ahmed and Ali, 2024</xref>), but different algorithms have different application effects in different research areas. Therefore, these five algorithms are selected in this paper for comparison.</p>
<p>
<xref ref-type="table" rid="T6">Table 6</xref> presents a comparison of the classification accuracy and F1 scores of different models before and after ST comprehensive sampling. Before ST sampling, the models achieved an accuracy of 50.56%&#x2013;72.12% and F1 scores of 36.01%&#x2013;62.1%. This was because the imbalanced dataset caused the models to overly rely on features from the majority class samples, resulting in low recognition ability for minority class samples and thus weakening the models&#x2019; ability to identify lithology.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Effect of ST integrated sampling on the performance of different models for lithology classification.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model</th>
<th align="center">Before ST sampling accuracy</th>
<th align="center">After ST sampling accuracy</th>
<th align="center">Accuracy improvement rate</th>
<th align="center">Before ST sampling F1 score</th>
<th align="center">After ST sampling<break/>F1 score</th>
<th align="center">F1 score improvement rate</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">LightGBM</td>
<td align="center">72.12%</td>
<td align="center">78.13%</td>
<td align="center">6.01%</td>
<td align="center">62.10%</td>
<td align="center">77.98%</td>
<td align="center">15.88%</td>
</tr>
<tr>
<td align="center">KNN (<xref ref-type="bibr" rid="B15">Guo et al., 2003</xref>)</td>
<td align="center">50.56%</td>
<td align="center">65.62%</td>
<td align="center">15.06%</td>
<td align="center">36.01%</td>
<td align="center">65.14%</td>
<td align="center">29.13%</td>
</tr>
<tr>
<td align="center">DT (<xref ref-type="bibr" rid="B27">Ren et al., 2023</xref>)</td>
<td align="center">55.39%</td>
<td align="center">60.68%</td>
<td align="center">5.29%</td>
<td align="center">46.16%</td>
<td align="center">60.25%</td>
<td align="center">14.09%</td>
</tr>
<tr>
<td align="center">GBDT (<xref ref-type="bibr" rid="B10">Dev and Eden, 2019</xref>)</td>
<td align="center">59.33%</td>
<td align="center">67.46%</td>
<td align="center">8.13%</td>
<td align="center">56.01%</td>
<td align="center">67.19%</td>
<td align="center">11.18%</td>
</tr>
<tr>
<td align="center">XGBoost (<xref ref-type="bibr" rid="B33">Sun et al., 2020b</xref>)</td>
<td align="center">60.00%</td>
<td align="center">73.91%</td>
<td align="center">13.91%</td>
<td align="center">58.31%</td>
<td align="center">75.27%</td>
<td align="center">16.96%</td>
</tr>
<tr>
<td align="center">RF (<xref ref-type="bibr" rid="B1">Ahmed and Ali, 2024</xref>)</td>
<td align="center">69.52%</td>
<td align="center">77.60%</td>
<td align="center">8.08%</td>
<td align="center">57.73%</td>
<td align="center">77.43%</td>
<td align="center">19.70%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>After ST comprehensive sampling, the dataset introduced new feature information, leading to a more balanced distribution of samples across various categories. The models could effectively extract features from minority class samples, resulting in an improvement in accuracy by 5.29%&#x2013;15.06% and F1 scores by 11.18%&#x2013;29.13%. The LightGBM model performed particularly well with an accuracy of 78.13% and an F1 score of 77.98%. ST comprehensive sampling effectively addressed the issue of dataset imbalance and significantly enhanced the lithological recognition performance of various classifiers, thereby increasing the models&#x2019; robustness and applicability.</p>
<p>Due to the overall poor classification performance of the model before ST composite sampling, the ST-ARO-LightGBM model was compared with different models established after sampling. <xref ref-type="fig" rid="F7">Figure 7</xref> shows the confusion matrices calculated for different models on the test set, visually reflecting the models&#x2019; ability to identify various lithologies. The horizontal axis is the true lithology of the sample, the vertical axis is the predicted lithology, and the integer represents the number of samples classified, and the percentage is the corresponding proportion. The darker the color on the main diagonal of the confusion matrix, the better the classification effect on lithology. The ST-ARO-LightGBM model demonstrated the best lithology classification performance, with recognition rates of over 90% for siltstone and detrital-bearing dolomite, and a resolution of 76.84% for cloud-bearing sandstone. However, the resolution for mudstone was relatively low at only 66.67%. This low resolution is due to the interference of logging feature ambiguities, which obscure the differences between mudstone and other lithologies. Other models also exhibited similar characteristics, with the lowest performance seen in the ST-DT model, which had a classification accuracy of only 43.43% for mudstone. This indicates that the proposed ST-ARO-LightGBM model can effectively reduce the interference of weak differences in logging features between different lithologies. To address the issue of low recognition accuracy for mudstone, future research should include more samples to explore the differences between mudstone and other lithologies.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Comparison of confusion matrix for lithology identification by different models. <bold>(A)</bold> ST-ARO-LightGBM. <bold>(B)</bold> ST-LightGBM. <bold>(C)</bold> ST-KNN. <bold>(D)</bold> ST-DT. <bold>(E)</bold> ST-GBDT. <bold>(F)</bold> ST-XGBoost. <bold>(G)</bold> ST-RF.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g007.tif"/>
</fig>
<p>
<xref ref-type="table" rid="T7">Table 7</xref> presents the accuracy, precision, recall, and F1 score of the ST-ARO-LightGBM model and six other models. Through comparative analysis, the ST-ARO-LightGBM model proposed in this study outperforms the other six models in all metrics, consistently achieving over 80%. Combined with <xref ref-type="table" rid="T7">Table 7</xref> and <xref ref-type="fig" rid="F7">Figure 7</xref>, the ST-ARO-LightGBM model demonstrates high accuracy and stability in predicting four different lithologies. Compared to traditional machine learning algorithms, it exhibits better robustness and application prospects.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Evaluation metrics table of lithology identification by model method after ST sampling.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center"/>
<th align="center">Accuracy (%)</th>
<th align="center">Precision (%)</th>
<th align="center">Recall (%)</th>
<th align="center">F1 Score (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">ST-ARO-LightGBM</td>
<td align="center">81.25</td>
<td align="center">81.01</td>
<td align="center">81.44</td>
<td align="center">81.07</td>
</tr>
<tr>
<td align="center">ST-LightGBM</td>
<td align="center">78.13</td>
<td align="center">78.02</td>
<td align="center">78.29</td>
<td align="center">77.98</td>
</tr>
<tr>
<td align="center">ST-KNN</td>
<td align="center">65.62</td>
<td align="center">65.22</td>
<td align="center">65.88</td>
<td align="center">65.14</td>
</tr>
<tr>
<td align="center">ST-DT</td>
<td align="center">60.68</td>
<td align="center">59.95</td>
<td align="center">60.83</td>
<td align="center">60.25</td>
</tr>
<tr>
<td align="center">ST-GBDT</td>
<td align="center">67.45</td>
<td align="center">67.13</td>
<td align="center">67.52</td>
<td align="center">67.19</td>
</tr>
<tr>
<td align="center">ST-XGBoost</td>
<td align="center">73.91</td>
<td align="center">75.37</td>
<td align="center">75.75</td>
<td align="center">75.27</td>
</tr>
<tr>
<td align="center">ST-RF</td>
<td align="center">77.6</td>
<td align="center">77.32</td>
<td align="center">77.76</td>
<td align="center">77.43</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The results show that the prediction result of integrated machine learning model (such as LightGBM) is better than that of single machine learning model (such as DT and KNN). In addition, the LightGBM model can achieve higher accuracy after being optimized by ARO algorithm, and the performance of machine learning model can be deeply explored, but the iteration time is increased. In the actual modeling process, attention should be paid to the balance between accuracy and computing power, and iteration can be ended in advance within the acceptable accuracy range to achieve the highest accuracy as possible.</p>
</sec>
<sec id="s5-2">
<title>5.2 Model application example</title>
<p>To verify the application of the ST-ARO-LightGBM model in the study area, predictions were made for well sections containing different lithologies, as shown in <xref ref-type="fig" rid="F8">Figures 8</xref>, <xref ref-type="fig" rid="F9">9</xref>. In these figures, the first panel displays depth, the second to fourth panels show input parameters for the model, the fifth panel presents lithological information from rock thin sections not used in model building, the sixth panel shows the lithology predictions and errors of the ST-ARO-LightGBM model, and the seventh to twelfth panels compare the lithology predictions and errors of other models.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Comparison of lithology of real thin section of rock in well A17X with the lithology identification results of the model proposed in this paper.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g008.tif"/>
</fig>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Comparison of lithology of real thin section of rock in well A32X with the lithology identification results of the model proposed in this paper.</p>
</caption>
<graphic xlink:href="feart-13-1540035-g009.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="F8">Figure 8</xref> illustrates the prediction results for well section A17X from 3,272 to 3,300 m, where mudstone and cloud-bearing sandstone are predominantly developed, with lesser occurrences of siltstone and detrital-bearing dolomite. Through comparison with actual rock thin section lithologies, the ST-ARO-LightGBM model significantly outperforms the six comparison models, accurately reflecting the complex interactions between different lithologies in the actual formations. Errors in the proposed model mainly occur in thin mudstone layers, consistent with the results in <xref ref-type="table" rid="T5">Table 5</xref> and <xref ref-type="fig" rid="F7">Figure 7</xref>.</p>
<p>
<xref ref-type="fig" rid="F9">Figure 9</xref> displays the prediction results for well section A32X from 3,724 to 3,736 m, where siltstone and mudstone are primarily developed. The predictions of the ST-ARO-LightGBM model closely match the actual rock thin section lithologies, whereas the comparison models generally capture the variation trends in lithology within the well section but with higher errors.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>Based on comprehensive experimental results on Permian shale reservoir lithology identification in Jimusaer, the following main conclusions are drawn:<list list-type="simple">
<list-item>
<p>(1) The Permian shale reservoir in Jimusaer primarily consists of mudstone, siltstone, cloud-bearing sandstone, and detrital-bearing dolomite. These lithologies coexist within the actual formations with similar logging response characteristics, posing challenges for traditional lithology identification methods.</p>
</list-item>
<list-item>
<p>(2) The ST composite sampling method effectively enhances the information of minority class samples while reducing redundant information from majority class samples, achieving balanced sample data. This process significantly improves the machine learning model&#x2019;s performance in lithology prediction tasks and enhances the model&#x2019;s classification performance and robustness.</p>
</list-item>
<list-item>
<p>(3) By integrating the multi-objective optimization strategy ARO algorithm with the ST-LightGBM model to establish the ST-ARO-LightGBM model, this study efficiently addresses the complex parameter adjustment issues of the ST-LightGBM model, optimizes the model structure, and enhances the lithology prediction capability and applicability of the model.</p>
</list-item>
</list>
</p>
<p>Although the model proposed in this paper can distinguish different lithologies of shale reservoirs to a certain extent, improve the unbalanced sample set, improve the lithology identification accuracy, and has certain universality, it will increase the calculation time cost due to the need for multiple iterations in the process of model establishment. In addition, the comprehensive analysis of multi-source and multi-modal data has become a research hotspot in the field of artificial intelligence. In future research, on the one hand, we can focus on improving the performance of the machine learning model itself, and on the other hand, we can combine various types of logs data for comprehensive evaluation and analysis.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The data analyzed in this study is subject to the following licenses/restrictions: The data sets in this study are restricted due to privacy concerns and are not publicly available. Requests to access these datasets should be directed to Cheng Feng, <email>fcvip0808@126.com</email>.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>XD: Investigation, Project administration, Resources, Supervision, Writing&#x2013;review and editing. JL: Conceptualization, Investigation, Methodology, Project administration, Supervision, Validation, Writing&#x2013;review and editing. JC: Data curation, Software, Validation, Visualization, Writing&#x2013;original draft. CF: Conceptualization, Data curation, Formal Analysis, Funding acquisition, Project administration, Resources, Software, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. National Natural Science Foundation of China (No. 42364007, 42004089), the Natural Science Foundation of Xinjiang Uygur Autonomous Region (No. 2021D01E22), the Innovative Outstanding Young Talents of Karamay, Key Research and Development Projects of Xinjiang Uygur Autonomous Region (2024B01016, 2024B01016-1, 2024B01016-3).</p>
</sec>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>Authors XD and JL were employed by PetroChina.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ahmed</surname>
<given-names>I. B.</given-names>
</name>
<name>
<surname>Ali</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Random forest and decision tree facies classification models for well log data of the mishrif formation from Basrah Oil Company, Southern Iraq</article-title>. <source>Iraqi Geol. J.</source> <volume>57</volume> (<issue>2E</issue>), <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.46717/igj.57.2E.2ms-2024-11-11</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Anazi</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Gates</surname>
<given-names>I. D.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>On the capability of support vector machines to classify lithology from well logs</article-title>. <source>Nat. Resour. Res.</source> <volume>19</volume>, <fpage>125</fpage>&#x2013;<lpage>139</lpage>. <pub-id pub-id-type="doi">10.1007/s11053-010-9118-9</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baisakhi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Rima</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Well log data analysis for lithology and fluid identification in Krishna-Godavari Basin, India</article-title>. <source>Arabian J. Geosciences</source> <volume>11</volume>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1007/s12517-018-3587-2</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bestagini</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Vincenzo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Stefano</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>A machine learning approach to facies classification using well logs</article-title>,&#x201d; in <source>Seg technical program expanded abstracts 2017</source>. <publisher-name>Society of Exploration Geophysicists</publisher-name>, <fpage>2137</fpage>&#x2013;<lpage>2142</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bressan</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>de Souza</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Girelli</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Junior</surname>
<given-names>F. C.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>Evaluation of machine learning methods for lithology classification using geophysical data</article-title>. <source>Comput. and Geosciences</source> <volume>139</volume>, <fpage>104475</fpage>. <pub-id pub-id-type="doi">10.1016/j.cageo.2020.104475</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bressan</surname>
<given-names>T. S.</given-names>
</name>
<name>
<surname>Kehl de Souza</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Girelli</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Junior</surname>
<given-names>F. C.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Evaluation of machine learning methods for lithology classification using geophysical data</article-title>. <source>Comput. and Geosciences</source> <volume>139</volume>, <fpage>104475</fpage>. <pub-id pub-id-type="doi">10.1016/j.cageo.2020.104475</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Deng</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Shan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zong</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Intelligent classification of volcanic rocks based on honey badger optimization algorithm enhanced Extreme gradient boosting tree model: a case study of hongche fault zone in Junggar Basin</article-title>. <source>Processes</source> <volume>12</volume> (<issue>2</issue>), <fpage>285</fpage>. <pub-id pub-id-type="doi">10.3390/pr12020285</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chioma</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Uko</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>Tamunobereton-ari</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Determination of lithology and pore-fluid of A reservoir in parts of Niger Delta using well-log data</article-title>. <source>J. Appl. Phys.</source> <volume>10</volume> (<issue>2</issue>), <fpage>71</fpage>&#x2013;<lpage>82</lpage>. <pub-id pub-id-type="doi">10.9790/4861-1002017182</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A real&#x2010;time lithological identification method based on SMOTE&#x2010;Tomek and ICSA optimization</article-title>. <source>Acta Geol. Sinica&#x2010;English Ed.</source> <volume>98</volume>, <fpage>518</fpage>&#x2013;<lpage>530</lpage>. <pub-id pub-id-type="doi">10.1111/1755-6724.15144</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dev</surname>
<given-names>V. A.</given-names>
</name>
<name>
<surname>Eden</surname>
<given-names>M. R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Formation lithology classification using scalable gradient boosted decision trees</article-title>. <source>Comput. and Chem. Eng.</source> <volume>128</volume>, <fpage>392</fpage>&#x2013;<lpage>404</lpage>. <pub-id pub-id-type="doi">10.1016/j.compchemeng.2019.06.001</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Devi</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Purkayastha</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Redundancy-driven modified Tomek-link based undersampling: a solution to class imbalance</article-title>. <source>Pattern Recognit. Lett.</source> <volume>93</volume>, <fpage>3</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1016/j.patrec.2016.10.006</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Prediction of vitrinite reflectance of shale oil reservoirs using nuclear magnetic resonance and conventional log data</article-title>. <source>Fuel</source> <volume>339</volume>, <fpage>127422</fpage>. <pub-id pub-id-type="doi">10.1016/j.fuel.2023.127422</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A novel method to estimate resistivity index of tight sandstone reservoirs using nuclear magnetic resonance logs</article-title>. <source>J. Nat. Gas Sci. Eng.</source> <volume>79</volume>, <fpage>103358</fpage>. <pub-id pub-id-type="doi">10.1016/j.jngse.2020.103358</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhong</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ling</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Determination of reservoir wettability based on resistivity index prediction from core and log data</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>205</volume>, <fpage>108842</fpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2021.108842</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bell</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Greer</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2003</year>). <source>KNN model-based approach in classification[C]//On the move to meaningful internet systems 2003: CoopIS, DOA, and ODBASE: OTM confederated international conferences, CoopIS, DOA, and ODBASE 2003, Catania, Sicily, Italy</source>. <publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>, <fpage>986</fpage>&#x2013;<lpage>996</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Prediction of igneous lithology and lithofacies based on ensemble learning with data optimization</article-title>. <source>Geophysics</source> <volume>89</volume> (<issue>2</issue>), <fpage>1</fpage>&#x2013;<lpage>JM11</lpage>. <pub-id pub-id-type="doi">10.1190/geo2022-0782.1</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cai</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Duan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Lithology identification of volcanic logging based on improved random forest</article-title>. <source>Sci. Technol. Eng.</source> <volume>23</volume> (<issue>09</issue>), <fpage>3696</fpage>&#x2013;<lpage>3704</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Imamverdiyev</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lyudmila</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Lithological facies classification using deep convolutional neural network</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>174</volume>, <fpage>216</fpage>&#x2013;<lpage>228</lpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2018.11.023</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ke</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Finley</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Lightgbm: a highly efficient gradient boosting decision tree</article-title>. <source>Adv. neural Inf. Process. Syst.</source>, <fpage>30</fpage>. <pub-id pub-id-type="doi">10.5555/3294996.3295074</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiaomin</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chengxu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A comprehensive machine learning model for lithology identification while drilling</article-title>. <source>Geoenergy Sci. Eng.</source> <volume>231</volume>, <fpage>212333</fpage>. <pub-id pub-id-type="doi">10.1016/j.geoen.2023.212333</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mitchell</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Monitoring lithology variations in drilled rock formations using NMR apparent magnetic susceptibility contrast</article-title>. <source>Appl. Magn. Reson.</source> <volume>51</volume> (<issue>3</issue>), <fpage>205</fpage>&#x2013;<lpage>219</lpage>. <pub-id pub-id-type="doi">10.1007/s00723-019-01157-1</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mou</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Comparison of three classical machine learning algorithms for lithology identification of volcanic rocks using well logging data</article-title>. <source>J. Jilin Univ. Sci. Ed.</source> <volume>51</volume> (<issue>03</issue>), <fpage>951</fpage>&#x2013;<lpage>956</lpage>. <pub-id pub-id-type="doi">10.13278/j.cnki.jjuese.20200210</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Passey</surname>
<given-names>Q. R.</given-names>
</name>
<name>
<surname>Bohacs</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Esch</surname>
<given-names>W. L.</given-names>
</name>
<name>
<surname>Klimentidis</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Sinha</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>From oil-prone source rock to gas-producing shale reservoir&#x2013;geologic and petrophysical characterization of unconventional shale-gas reservoirs</article-title>,&#x201d; in <source>SPE International oil and gas Conference and exhibition in China. SPE</source>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pereira</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Costa</surname>
<given-names>Y. M. G.</given-names>
</name>
<name>
<surname>Silla</surname>
<given-names>Jr C. N.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>MLTL: a multi-label approach for the Tomek Link undersampling algorithm</article-title>. <source>Neurocomputing</source> <volume>383</volume>, <fpage>95</fpage>&#x2013;<lpage>105</lpage>. <pub-id pub-id-type="doi">10.1016/j.neucom.2019.11.076</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prabowo</surname>
<given-names>U. N.</given-names>
</name>
<name>
<surname>Ferdiyan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Raharjo</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Sehah</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Candra</surname>
<given-names>A. D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Comparison of facies estimation using support vector machine (SVM) and K-nearest neighbor (KNN) algorithm based on well log data</article-title>. <source>Aceh Int. J. Sci. Technol.</source> <volume>12</volume> (<issue>2</issue>), <fpage>246</fpage>&#x2013;<lpage>253</lpage>. <pub-id pub-id-type="doi">10.13170/aijst.12.2.28428</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Probst</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Anne-Laure</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Bernd</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Tunability: importance of hyperparameters of machine learning algorithms</article-title>. <source>J. Mach. Learn. Res.</source> <volume>20</volume> (<issue>53</issue>), <fpage>1</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1802.09596</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Lithology identification using principal component analysis and particle swarm optimization fuzzy decision tree</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>220</volume>, <fpage>111233</fpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2022.111233</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saporetti</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Goliatt</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pereira</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Neural network boosted with differential evolution for lithology identification based on well logs information</article-title>. <source>Earth Sci. Inf.</source> <volume>14</volume>, <fpage>133</fpage>&#x2013;<lpage>140</lpage>. <pub-id pub-id-type="doi">10.1007/s12145-020-00533-x</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saporetti</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Leonardo Goliatt da</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Egberto</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A lithology identification approach based on machine learning with evolutionary parameter tuning</article-title>. <source>IEEE Geoscience Remote Sens. Lett.</source> <volume>16</volume> (<issue>12</issue>), <fpage>1819</fpage>&#x2013;<lpage>1823</lpage>. <pub-id pub-id-type="doi">10.1109/lgrs.2019.2911473</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sebtosheikh</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Motafakkerfard</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Riahi</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Moradi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sabety</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Support vector machine method, a new technique for lithology prediction in an Iranian heterogeneous carbonate reservoir using petrophysical well logs</article-title>. <source>Carbonates evaporites</source> <volume>30</volume>, <fpage>59</fpage>&#x2013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.1007/s13146-014-0199-0</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Maheswar</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Machine learning in the classification of lithology using downhole NMR data of the NGHP-02 expedition in the Krishna-Godavari offshore Basin, India</article-title>. <source>Mar. Petroleum Geol.</source> <volume>135</volume>, <fpage>105443</fpage>. <pub-id pub-id-type="doi">10.1016/j.marpetgeo.2021.105443</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ji</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xiang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020a</year>). <article-title>Identification of complex carbonate lithology by logging based on XGBoost algorithm</article-title>. <source>Lithol. Reserv.</source> <volume>32</volume> (<issue>04</issue>), <fpage>98</fpage>&#x2013;<lpage>106</lpage>. <pub-id pub-id-type="doi">10.12108/yxyqc.20200410</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sun</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020b</year>). <article-title>A data-driven approach for lithology identification based on parameter-optimized ensemble learning</article-title>. <source>Energies</source> <volume>13</volume> (<issue>15</issue>), <fpage>3903</fpage>. <pub-id pub-id-type="doi">10.3390/en13153903</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Thongsamea</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Kanitpanyacharoena</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Chuangsuwanich</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Lithological classification from well logs using machine learning algorithms</article-title>. <source>Bull. Earth Sci. Thail.</source> <volume>10</volume> (<issue>1</issue>), <fpage>31</fpage>&#x2013;<lpage>43</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mirjalili</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Artificial rabbits optimization: a new bio-inspired meta-heuristic algorithm for solving engineering optimization problems</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>114</volume>, <fpage>105082</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2022.105082</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Nie</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wan</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Identification of complex carbonate lithology based on random forest algorithm</article-title>. <source>Chin. J. Eng. Geophys.</source> <volume>17</volume> (<issue>05</issue>), <fpage>550</fpage>&#x2013;<lpage>558</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.1672-7940.2020.05.003</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiong</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Response of well logging and &#x201c;sweet spot&#x201d; rapid evaluation technology for shale oil in the Lucaogou Formation of Jimsar Sag</article-title>. <source>Special Oil and Gas Reservoirs</source> <volume>30</volume> (<issue>04</issue>), <fpage>35</fpage>&#x2013;<lpage>43</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.1006-6535.2023.04.005</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zha</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Characteristics and classification evaluation of shale oil reservoir of the Lucaogou Formation in Jimsa</source>. <comment>(Dissertation)</comment>. <publisher-name>Chongqing University of Science and Technology</publisher-name>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Progress in China&#x27;s unconventional oil and gas exploration and development and theoretical technologies</article-title>. <source>Geol. Rev.</source> <volume>89</volume> (<issue>06</issue>), <fpage>979</fpage>&#x2013;<lpage>1007</lpage>. <pub-id pub-id-type="doi">10.1111/1755-6724.12491</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>