<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Earth Sci.</journal-id>
<journal-title>Frontiers in Earth Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Earth Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-6463</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1601363</article-id>
<article-id pub-id-type="doi">10.3389/feart.2025.1601363</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Earth Science</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Data-driven intelligent productivity prediction model for horizontal fracture stimulation</article-title>
<alt-title alt-title-type="left-running-head">Li et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/feart.2025.1601363">10.3389/feart.2025.1601363</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Li</surname>
<given-names>Qian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2854388/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Sui</surname>
<given-names>Yiyong</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Luo</surname>
<given-names>Mengying</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Guan</surname>
<given-names>Bin</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing &#x2013; review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Lu</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Yuan</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Petroleum Engineering</institution>, <institution>China University of Petroleum (East China)</institution>, <addr-line>Qingdao</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Petroleum Development Center of ShengLi Oilfield</institution>, <addr-line>Dongying</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Engineering Technology Department of Daqing Oilfield Limited Company</institution>, <addr-line>Daqing</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Exploration and Development Research Institute of Dagang Oilfield Company</institution>, <addr-line>Tianjin</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Tianjin Branch of CNPC Logging</institution>, <addr-line>Tianjin</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1902602/overview">Xin Sun</ext-link>, Sinopec Matrix Co., LTD., China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/841061/overview">Fuqiong Huang</ext-link>, China Earthquake Networks Center, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1828966/overview">Yingkun Fu</ext-link>, University of Alberta, Canada</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3018629/overview">Fengjiao Zhang</ext-link>, China University of Petroleum, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Yiyong Sui, <email>suiyy@126.com</email>, <email>suiyy@upc.edu.cn</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>01</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1601363</elocation-id>
<history>
<date date-type="received">
<day>27</day>
<month>03</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>30</day>
<month>06</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Li, Sui, Luo, Guan, Liu and Zhao.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Li, Sui, Luo, Guan, Liu and Zhao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Traditional methods for predicting post-fracturing productivity in horizontal fractures primarily use fracture and formation parameters for calculations. Complex fracture data are difficult to obtain, and these methods do not consider the effects of displacement mechanisms, fracturing techniques, or time factors on post-fracturing productivity. To address the limitations and shortcomings of existing post-fracturing performance prediction methods for horizontal fractures, a horizontal fracture well productivity prediction model was established by combining physical mechanisms with data-driven approaches. First, based on physical mechanisms, factors influencing well productivity were selected from reservoir properties and fracturing operations. Second, relevant characteristic parameters were chosen from geological conditions, production characteristics, and fracturing techniques to perform clustering analysis on fracturing intervals in the data sample. Intervals with similar multidimensional physical features were grouped into the same category. Under the assumption of similar characteristics and mechanisms, correlation analysis was conducted for each fracturing interval category to identify the dominant controlling factors affecting post-fracturing productivity in each reservoir type. Machine learning algorithms were used to establish intelligent models describing the relationships between post-fracturing production enhancement effects, dominant factors, and production time for each reservoir category. Finally, during fracturing design, the optimal productivity prediction model was matched to each interval based on its characteristics to predict post-fracturing productivity. Additionally, the influence patterns of proppant volume on well productivity were comprehensively analyzed to optimize reasonable proppant volumes for different wells and intervals. Field validation showed that the productivity prediction model achieved an average error of 7.06%, providing a basis for horizontal fracture engineering design and achieving cost reduction and efficiency improvement in oilfield development.</p>
</abstract>
<kwd-group>
<kwd>data-driven</kwd>
<kwd>horizontal fracture fracturing</kwd>
<kwd>productivity optimization</kwd>
<kwd>application</kwd>
<kwd>controlling factors</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Solid Earth Geophysics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Fracturing of Vertical wells in shallow oil reservoirs has the characteristics of small scale, low cost, and good effectiveness, is one of the most commonly used methods to increase production. In shallow oil reservoirs, the relatively low vertical stress coupled with a high horizontal <italic>in-situ</italic> stress dominance results in a propensity for hydraulic fractures to propagate horizontally. As the cornerstone of fracturing design optimization, productivity forecasting for hydraulically fractured wells encompasses dual methodologies: stimulation ratio quantification and production profile prediction. Conventional evaluation of horizontal fracture efficacy predominantly relies on analytical solutions for stimulation ratio (SR) computation. Since <xref ref-type="bibr" rid="B7">McGuire and Sikora (1960)</xref> pioneered the methodology for predicting productivity of vertically fractured wells via stimulation ratio in 1960, numerous scholars worldwide have conducted extensive studies under diverse conditions to investigate the relationships between stimulation ratios and fracture geometry dimensions coupled with conductivity. <xref ref-type="bibr" rid="B10">Prats (1961)</xref> proposed a methodology for calculating the SR under steady-state flow conditions in 1961. Cui Disheng (<xref ref-type="bibr" rid="B2">Choi, 1986</xref>) developed a comprehensive SR calculation framework incorporating multiple contributing factors: fracture stimulation effects, formation damage mitigation, drainage area configuration, and well placement optimization, thereby advancing a more holistic predictive model for post-fracturing productivity enhancement. <xref ref-type="bibr" rid="B5">Liang and Zhao (2019)</xref> established a correlation between Estimated Ultimate Recovery (EUR) and 10 key production-influencing factors for fractured horizontal wells using Random Forest algorithms. <xref ref-type="bibr" rid="B11">Raymond and Binder (1967)</xref> analytically derived stimulation ratios for a centrally located fractured well within a circular drainage area under pseudo-steady state flow conditions. <xref ref-type="bibr" rid="B16">Zhang (2020)</xref> developed a COMSOL-based predictive model quantifying stimulation ratios and liquid production rates through systematic analysis of reservoir properties (permeability, porosity), fracture parameters (length, conductivity), and operational variables (flow rate, bottom hole pressure), explicitly accounting for parameter sensitivity and cross-correlation effects. <xref ref-type="bibr" rid="B6">Ma (2022)</xref> formulated a semi-analytical single-well model for fractured horizontal wells in heterogeneous tight oil reservoirs, leveraging Laplace-space Green&#x2019;s functions for rectangular domains. This model specifically evaluates fracture penetration ratio and permeability contrast impacts on post-fracturing productivity. <xref ref-type="bibr" rid="B8">Mohaghegh et al. (2017)</xref> implemented a neuro-fuzzy framework combining fuzzy clustering for well typology classification, Key Performance Indicator analysis for dominant factor identification, and Artificial Neural Networks integrating geological (TOC, brittleness index), completion (stage spacing), and stimulation (proppant intensity) parameters to predict early-phase production. <xref ref-type="bibr" rid="B9">Pan et al. (2018)</xref> applied grey correlation analysis to identify critical productivity drivers&#x2014;proppant volume, net pay thickness, total injected fluid volume, and stimulation stages&#x2014;subsequently constructing a multiple linear regression model for initial production forecasting in tight oil horizontal wells. <xref ref-type="bibr" rid="B12">Wang and Chen (2019)</xref> proposed machine learning-driven productivity prediction models for hydraulically fractured horizontal wells by integrating Artificial Neural Networks and Support Vector Machines (SVM). <xref ref-type="bibr" rid="B15">Yan et al. (2021)</xref> developed predictive frameworks for post-fracturing production in tight sandstone reservoirs under limited historical datasets, employing an ensemble of Elastic Net regression, Decision Trees, and SVM to identify salient parameters from multi-domain fracture-influencing factors. <xref ref-type="bibr" rid="B13">Tang and Wang (2023)</xref> constructed an XGBoost-based productivity forecasting model for 267 horizontal wells in the Sulige Gas Field, leveraging petrophysical features (gas saturation, porosity) and engineering variables (total proppant volume, cluster spacing) as model inputs to quantify fracture-reservoir interactions.</p>
<p>In summary, analytical formulas for post-fracturing performance prediction inherently require fracture parameters as inputs. However, shallow reservoir fracturing operations are typically low-cost and small-scale, making fracture parameter acquisition challenging. Furthermore, with advancements in multi-fracture stimulation technologies, critical variables such as fracture count are absent in conventional analytical models. These formulas yield static productivity estimates, whereas actual post-fracturing production exhibits dynamic temporal decline behavior. Consequently, analytical approaches demonstrate limited applicability in modern fracturing evaluation. Existing data-driven methodologies predominantly adopt well-centric modeling frameworks while neglecting the impacts of diverse displacement mechanisms (e.g., water flooding, polymer flooding, ASP flooding) on stimulated productivity. In shallow reservoirs undergoing various enhanced oil recovery processes, distinct displacement physics&#x2013;including viscosity modification (polymer), interfacial tension reduction (surfactant), and mobility control (ASP) &#x2013; differentially influence fracture-reservoir interactions. Therefore, developing displacement mechanism-specific data-driven productivity prediction models for targeted fracturing intervals shows significant potential to enhance fracturing design accuracy and optimize cost-benefit ratios.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methodology</title>
<sec id="s2-1">
<title>2.1 Main technical principles</title>
<p>Post-fracturing performance varies significantly across different fracturing intervals, stimulation techniques, and displacement mechanisms due to distinct underlying physical mechanisms. The workflow involves three key steps: (as shown in <xref ref-type="fig" rid="F1">Figure 1</xref>):<list list-type="simple">
<list-item>
<p>1. Data Collection and preprocessing: Sample data from wells and target fracturing intervals are collected and preprocessed.</p>
</list-item>
<list-item>
<p>2. Fracturing layer clustering: fracturing intervals are grouped into clusters based on multidimensional features, including geological conditions production characteristics, and stimulation parameters Clusters are formed under the assumption that intervals within the same group share identical attributes, mechanisms, and post-fracturing productivity enhancement patterns.</p>
</list-item>
<list-item>
<p>3. Model Development: For each cluster, dominant factors governing horizontal fracture performance are analyzed. Tailored productivity prediction models are then established for horizontal fracture-dominated intervals, achieving higher accuracy and efficiency for specific geological and operational categories.</p>
</list-item>
</list>
</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Modeling schematic of the fractured well capacity prediction model.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g001.tif">
<alt-text content-type="machine-generated">Flowchart depicting a process starting with &#x22;Fractured Well Sample Data&#x22; undergoing clustering into multiple categories (0 to k). Each category is then analyzed for correlation, leading to &#x22;Master Factors&#x22; for each category. These factors undergo modeling to produce models (M&#x2080; to M&#x2096;).</alt-text>
</graphic>
</fig>
<p>During fracturing design, the target interval is classified using a clustering model based on its characteristic features. The classified interval category is then matched with the optimal productivity prediction model tailored for that specific interval type. By activating the selected model and inputting relevant interval data and stimulation parameters, the post-fracturing production behavior&#x2014;including production trends and decline patterns&#x2014;can be predicted for the target interval. As shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Schematic diagram of post-fracturing capacity intelligence model application.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g002.tif">
<alt-text content-type="machine-generated">Flowchart depicting the process of classifying fractured well sample data. The data is grouped through a clustering process into categories labeled 0, 1, up to k. New wells data is processed using model Mj, which outputs the new well post-pressing effect.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2-2">
<title>2.2 Data collection and preprocessing</title>
<p>Based on the physical mechanisms of fracturing-induced productivity enhancement, 12 characteristic parameters were selected, encompassing pre-fracturing geological parameters, production parameters, and fracturing operational parameters, along with two additional features: displacement mechanism and fracturing technique type. The displacement mechanisms primarily include water flooding, polymer flooding, and ASP flooding, while fracturing techniques are categorized into conventional fracturing, multi-fracture stimulation, and selective zonal fracturing. A dataset comprising 5,268 fractured well samples was established, as detailed in <xref ref-type="table" rid="T1">Table 1</xref>. The distribution of sample data across these categories is presented in <xref ref-type="table" rid="T2">Table 2</xref>. The dataset structured enables physics-informed machine learning while maintaining operational reality constraints.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Data composition of the data sample set.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Displacement method</th>
<th align="center">Fracturing wells / well</th>
<th align="center">Fracturing technology</th>
<th align="center">Fracturing wells / well</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Water flooding</td>
<td align="center">1878</td>
<td align="center">Conventional fracturing</td>
<td align="center">921</td>
</tr>
<tr>
<td align="center">Polymer flooding</td>
<td align="center">2379</td>
<td align="center">Multi-Fracture Fracturing</td>
<td align="center">3267</td>
</tr>
<tr>
<td align="center">Alkaline-Surfactant-Polymer</td>
<td align="center">1011</td>
<td align="center">Selective Fracturing</td>
<td align="center">1080</td>
</tr>
<tr>
<td align="center">Total</td>
<td align="center">5268</td>
<td align="center">Total</td>
<td align="center">5268</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Data range of dataset.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Data range</th>
<th align="center">Present formation<break/> pressure /MPa</th>
<th align="center">Well spacing /m</th>
<th align="center">Depth in the middle<break/> of the oil layer /m</th>
<th align="center">Sandstone thickness /m</th>
<th align="center">Effective thickness /m</th>
<th align="center">Porosity /%</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Minimum value</td>
<td align="center">5.21</td>
<td align="center">100</td>
<td align="center">798</td>
<td align="center">0.2</td>
<td align="center">0.1</td>
<td align="center">20</td>
</tr>
<tr>
<td align="center">Maximum values</td>
<td align="center">20.48</td>
<td align="center">300</td>
<td align="center">1192</td>
<td align="center">19.3</td>
<td align="center">12.4</td>
<td align="center">53</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="center">Permeability<break/> /&#x3bc;m<sup>2</sup>
</th>
<th align="center">Fracture uncot<break/> /strip</th>
<th align="center">Total fracturing fluid<break/> volume /m<sup>3</sup>
</th>
<th align="center">Proppant<break/> volume /m<sup>3</sup>
</th>
<th align="center">Pre-fracture water<break/> content /%</th>
<th align="center">Daily oil production from<break/> small layer before<break/> fracking /t</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Minimum value</td>
<td align="center">0.004</td>
<td align="center">1</td>
<td align="center">32</td>
<td align="center">4.5</td>
<td align="center">82</td>
<td align="center">0.001</td>
</tr>
<tr>
<td align="center">Maximum values</td>
<td align="center">1.5</td>
<td align="center">2</td>
<td align="center">111</td>
<td align="center">120</td>
<td align="center">99</td>
<td align="center">8.79</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>For clustering analysis of fracturing intervals, numerical data are required. In addition to standard preprocessing steps such as normalization, the original dataset contains two categorical/textual variables&#x2014;displacement mechanisms and fracturing techniques&#x2014;which were converted into numerical representations using dummy variable encoding (<xref ref-type="bibr" rid="B4">Jin et al., 2023</xref>). For displacement mechanisms, ASP flooding was selected as the reference category, with water flooding and polymer flooding encoded as &#x201c;01&#x201d; and &#x201c;10,&#x201d; respectively. For fracturing techniques, selective zonal fracturing served as the reference category, while conventional fracturing and multi-fracture stimulation were encoded as &#x201c;01&#x201d; and &#x201c;10.&#x201d; The dataset before and after transformation is presented in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Dummy variable coded transformational relationships between the replacement method and the fracturing process.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Data parameters</th>
<th align="center">Data categories before conversion</th>
<th align="center">Post-conversion data</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="3" align="center">Displacement method</td>
<td align="center">Alkaline-Surfactant-Polymer</td>
<td align="center">00</td>
</tr>
<tr>
<td align="center">Polymer flooding</td>
<td align="center">01</td>
</tr>
<tr>
<td align="center">Water flooding</td>
<td align="center">10</td>
</tr>
<tr>
<td rowspan="3" align="center">Fracturing technology</td>
<td align="center">Selective Fracturing</td>
<td align="center">00</td>
</tr>
<tr>
<td align="center">Conventional fracturing</td>
<td align="center">01</td>
</tr>
<tr>
<td align="center">Multi-Fracture Fracturing</td>
<td align="center">10</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-3">
<title>2.3 Cluster analysis of fractured intervals</title>
<p>The dataset is partitioned into subsets of similar characteristics based on the similarity between different intervals. Each subset contains samples with closely aligned properties, while maintaining distinct differences between subsets. This approach facilitates a comprehensive understanding of key information patterns within the fracturing well data. The K-means clustering algorithm (<xref ref-type="bibr" rid="B14">Wang, 2018</xref>; <xref ref-type="bibr" rid="B1">Chen and Xiao, 2004</xref>) &#x2014; an iterative analytical method&#x2014;provides interpretable results where cluster centroids represent the characteristic attributes of each group. As clustering outcomes critically depend on the selection of cluster numbers (k), determining the optimal k-value is pivotal, particularly when fracturing interval categories lack predefined definitions.</p>
<p>For this analysis, k-values ranging from 2 to 13 were systematically tested on the fracturing well dataset. Clustering performance was evaluated using the silhouette coefficient metric, which quantifies both intra-cluster cohesion (a(i)) and inter-cluster separation (b(i)) for each k-value. The silhouette score s(i) &#x2014; calculated as:<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi mathvariant="bold-italic">s</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">max</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">a</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="bold-italic">b</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi mathvariant="bold-italic">i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Serves as a composite evaluation criterion. Higher s(i) values indicate superior clustering configurations, with the k-value yielding the maximum s(i) identified as the optimal classification scheme for multidimensional fracturing interval features.</p>
<p>As shown in <xref ref-type="fig" rid="F3">Figure 3</xref>, the silhouette coefficient initially increases with the number of clusters (k) and gradually plateaus. The maximum silhouette coefficient is achieved at k &#x3d; 9, indicating optimal clustering results. Therefore, the fracturing intervals in the dataset exhibit the best clustering performance when classified into 9 categories.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Evaluation of Clustering Results. <bold>(A)</bold> The silhouette coefficient score situation for each cluster <bold>(B)</bold> Silhouette coefficient at k &#x3d; 9.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g003.tif">
<alt-text content-type="machine-generated">Graph panel (A) depicts the silhouette coefficient versus the number of clusters, showing an increase in the coefficient peaking at nine clusters, marked by a red dashed line. Graph panel (B) is a silhouette plot for cluster analysis, displaying varied cluster labels with silhouette coefficients, and indicating a mean coefficient of 0.862, highlighted by a red dashed line.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2-4">
<title>2.4 Principal controlling factor analysis</title>
<p>The Maximal Information Coefficient (MIC) is a nonparametric statistical method rooted in mutual information theory, designed to quantify association strength between variables, particularly adept at capturing complex linear and nonlinear relationships in high-dimensional data. Its core principle involves dynamically partitioning data grids to compute the maximum mutual information across varying resolutions. Compared to Pearson&#x2019;s correlation coefficient, which only identifies linear associations, MIC demonstrates significantly enhanced sensitivity to nonlinear patterns such as exponential, periodic, and piecewise relationships, while maintaining robustness against noise and outliers.</p>
<p>In hydraulic fracturing engineering, nonlinear characteristics frequently govern interactions between reservoir parameters (porosity <italic>&#x3d5;</italic>, permeability <italic>k</italic>), operational parameters (proppant volume m, fracture count <italic>N</italic>), and productivity. Certain parameters exhibit threshold effects on productivity enhancement&#x2014;where exceeding critical values leads to stabilized stimulation effects&#x2014;while others follow power-law relationships with production outcomes. Traditional regression models struggle to characterize such complexities, whereas MIC enables precise identification of dominant factors through global optimization of variable association patterns.</p>
<p>Based on geomechanical and flow theory, the following multidimensional parameters were analyzed:</p>
<p>Geological Parameters:<list list-type="simple">
<list-item>
<p>Porosity (<italic>&#x3d5;</italic>)</p>
</list-item>
<list-item>
<p>Permeability (<italic>k</italic>)</p>
</list-item>
<list-item>
<p>Sandstone thickness (<italic>h</italic>)</p>
</list-item>
<list-item>
<p>Reservoir mid-depth (<italic>D</italic>)</p>
</list-item>
</list>
</p>
<p>Operational Parameters:<list list-type="simple">
<list-item>
<p>Proppant volume (<italic>m</italic>)</p>
</list-item>
<list-item>
<p>Fracture count (<italic>N</italic>)</p>
</list-item>
</list>
</p>
<p>Dynamic Parameters:<list list-type="simple">
<list-item>
<p>Current reservoir pressure (<italic>P</italic>)</p>
</list-item>
<list-item>
<p>Well spacing (<italic>L</italic>)</p>
</list-item>
<list-item>
<p>Pre-fracturing water cut (<italic>f</italic>)</p>
</list-item>
</list>
</p>
<p>Target Variable:<list list-type="simple">
<list-item>
<p>Post-fracturing productivity (<italic>Q</italic>)</p>
</list-item>
</list>
</p>
<p>Z-score standardization applied to eliminate dimensional heterogeneity. For each variable pair (X<sub>
<italic>i</italic>
</sub>, Q), dynamic grid partitioning was performed in 2D space. Mutual information maxima were computed across grid resolutions. Normalized MIC values (0 &#x2264; MIC &#x2264;1) were derived, characteristics with a MIC value greater than 0.4 can be considered as primary control factors, with results visualized in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Correlation between main control factors and production capacity in different intervals. <bold>(A)</bold> Model 0 <bold>(B)</bold> Model 1 <bold>(C)</bold> Model 2 <bold>(D)</bold> Model 3 <bold>(E)</bold> Model 4 <bold>(F)</bold> Model 5 <bold>(G)</bold> Model 6 <bold>(H)</bold> Model 7 <bold>(I)</bold> Model 8.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g004.tif">
<alt-text content-type="machine-generated">Nine bar charts display the Maximal Information Coefficient (MIC) for parameters labeled P, L, D, h, &#x3C6;, k, m, N, and f. Each chart, labeled A through I, shows varying MIC values with a common pattern of high and low values across parameters. The MIC ranges from about 0.0 to 0.8, with differing heights for each parameter in individual charts. Green and blue bars represent different data sets or variables within each chart.</alt-text>
</graphic>
</fig>
<p>The results demonstrate that distinct geological characteristics and microscale mechanistic variations across reservoir categories lead to divergent macro-scale dominant factors governing post-fracturing productivity within each fracturing interval type.</p>
</sec>
<sec id="s2-5">
<title>2.5 Intelligent prediction modeling development</title>
<p>The training dataset for each category of fractured well productivity is denoted as D &#x3d; {(X<sub>i</sub>, Q<sub>i</sub>)} (i &#x3d; 1,2,3, &#x2026; ,n), where X<sub>i</sub>&#x2208;Rp represents the multidimensional feature vector determined by dominant governing factors, and Q<sub>i</sub>&#x2208;R corresponds to the post-fracturing productivity. The dataset D was split into training and testing sets in an 8:2 ratio. Machine learning was performed on the training set to develop post-fracturing productivity prediction models for each fracturing interval category using Gradient Boosted Regression Trees (GBRT), Random Forests (RF), and Bagging. The post-fracturing productivity prediction models were optimized after comparing and analyzing the evaluation metrics.</p>
<sec id="s2-5-1">
<title>2.5.1 Random forest models for post-fracturing capacity prediction</title>
<p>The Random forest algorithm is used to establish a post-fracturing capacity prediction model, and the schematic diagram of the Random forest regression algorithm is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. The maximum values of model tree depth, number of trees and corresponding R<sup>2</sup> coefficients of determination for the random forest model for capacity prediction are detailed in <xref ref-type="table" rid="T4">Table 4</xref>. The average value of the R<sup>2</sup> coefficient of determination of the nine types of fracturing well production capacity prediction models established by the random forest regression algorithm is 0.85 for the test set and 0.97 for the training set, which is a difference of 0.12.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Process of random forest training.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g005.tif">
<alt-text content-type="machine-generated">Flowchart illustrating the process of a random forest model for predicting post-fracturing well capacity. It starts with a training set of production capacity data, sampled randomly into subsets. Each subset generates a decision tree. The trees' outputs are integrated through voting to form the final prediction model.</alt-text>
</graphic>
</fig>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Tree depth, tree number and maximum R<sup>2</sup> determination coefficient of 9 types of fracturing well productivity prediction models (Random Forest).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model categories</th>
<th align="center">0</th>
<th align="center">1</th>
<th align="center">2</th>
<th align="center">3</th>
<th align="center">4</th>
<th align="center">5</th>
<th align="center">6</th>
<th align="center">7</th>
<th align="center">8</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Regression tree depth/layer</td>
<td align="center">15</td>
<td align="center">19</td>
<td align="center">18</td>
<td align="center">19</td>
<td align="center">17</td>
<td align="center">16</td>
<td align="center">15</td>
<td align="center">19</td>
<td align="center">16</td>
</tr>
<tr>
<td align="center">Number of regression trees/tree</td>
<td align="center">850</td>
<td align="center">100</td>
<td align="center">850</td>
<td align="center">950</td>
<td align="center">150</td>
<td align="center">350</td>
<td align="center">150</td>
<td align="center">900</td>
<td align="center">100</td>
</tr>
<tr>
<td align="center">Test set coefficient of determinationR<sup>2</sup>
</td>
<td align="center">0.90</td>
<td align="center">0.82</td>
<td align="center">0.83</td>
<td align="center">0.85</td>
<td align="center">0.89</td>
<td align="center">0.84</td>
<td align="center">0.83</td>
<td align="center">0.86</td>
<td align="center">0.82</td>
</tr>
<tr>
<td align="center">Training set coefficient of determinationR<sup>2</sup>
</td>
<td align="center">0.97</td>
<td align="center">0.96</td>
<td align="center">0.99</td>
<td align="center">0.97</td>
<td align="center">0.98</td>
<td align="center">0.98</td>
<td align="center">0.99</td>
<td align="center">0.96</td>
<td align="center">0.97</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-5-2">
<title>2.5.2 Bagging models for post-fracturing capacity prediction</title>
<p>The Bagging algorithm is used to establish a post-fracturing capacity prediction model, and the schematic diagram of the Bagging regression algorithm is shown in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Principle diagram of Bagging regression algorithm.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g006.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a process divided into training and forecasting stages. In the training stage, a dataset is split into sample subsets, each fed into separate models. The models form a composite model. In the forecasting stage, the composite model processes projection set data, distributing it across individual models, which then produce combined projected results.</alt-text>
</graphic>
</fig>
<p>The maximum values of model tree depth, number of trees and corresponding R<sup>2</sup> coefficients of determination for the Bagging model for capacity prediction are detailed in <xref ref-type="table" rid="T5">Table 5</xref>. As shown in <xref ref-type="table" rid="T5">Table 5</xref>, the R<sup>2</sup> coefficient of determination of the nine types of fracturing well production capacity prediction models established by Bagging regression algorithm is 0.87 on average for the test set and 0.95 on average for the training set, with an average difference of 0.09 between the two.</p>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Tree depth, tree number and maximum R<sup>2</sup> determination coefficient of 9 types of fracturing well productivity prediction models (Bagging).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model categories</th>
<th align="center">0</th>
<th align="center">1</th>
<th align="center">2</th>
<th align="center">3</th>
<th align="center">4</th>
<th align="center">5</th>
<th align="center">6</th>
<th align="center">7</th>
<th align="center">8</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Regression tree depth/layer</td>
<td align="center">19</td>
<td align="center">19</td>
<td align="center">20</td>
<td align="center">20</td>
<td align="center">18</td>
<td align="center">20</td>
<td align="center">19</td>
<td align="center">19</td>
<td align="center">20</td>
</tr>
<tr>
<td align="center">Number of regression trees/tree</td>
<td align="center">200</td>
<td align="center">100</td>
<td align="center">800</td>
<td align="center">1000</td>
<td align="center">200</td>
<td align="center">100</td>
<td align="center">350</td>
<td align="center">600</td>
<td align="center">750</td>
</tr>
<tr>
<td align="center">Test set coefficient of determinationR<sup>2</sup>
</td>
<td align="center">0.90</td>
<td align="center">0.91</td>
<td align="center">0.85</td>
<td align="center">0.89</td>
<td align="center">0.91</td>
<td align="center">0.85</td>
<td align="center">0.85</td>
<td align="center">0.88</td>
<td align="center">0.82</td>
</tr>
<tr>
<td align="center">Training set coefficient of determinationR<sup>2</sup>
</td>
<td align="center">0.98</td>
<td align="center">0.94</td>
<td align="center">0.97</td>
<td align="center">0.96</td>
<td align="center">0.98</td>
<td align="center">0.95</td>
<td align="center">0.96</td>
<td align="center">0.95</td>
<td align="center">0.94</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s2-5-3">
<title>2.5.3 GBRT models for post-fracturing capacity prediction</title>
<p>The GBRT algorithm is used to establish a post-fracturing capacity prediction model, and the schematic diagram of the GBRT regression algorithm is shown in <xref ref-type="fig" rid="F7">Figure 7</xref>. During the construction of the productivity prediction models, key parameters considered included tree depth and number of trees. A Bayesian optimization approach was employed to determine the hyperparameters of the nine GBRT models for different fracturing interval categories, as detailed in <xref ref-type="table" rid="T6">Table 6</xref>.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Establishment process of post fracturing production capacity prediction model.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g007.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a training process for forecasting post-fracturing wells production capacity. It involves an initial regression tree model, fracturing capacity forecasts, and negative gradient loss functions. The process iterates through multiple models, denoted as models one to m, aggregating results into a final regression value.</alt-text>
</graphic>
</fig>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Maximum values of model tree depth, number of trees, and corresponding R<sup>2</sup> coefficients of determination for 9 types of fractured well production capacity prediction models (GBRT).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model categories</th>
<th align="center">0</th>
<th align="center">1</th>
<th align="center">2</th>
<th align="center">3</th>
<th align="center">4</th>
<th align="center">5</th>
<th align="center">6</th>
<th align="center">7</th>
<th align="center">8</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Regression tree depth/layer</td>
<td align="center">7</td>
<td align="center">8</td>
<td align="center">10</td>
<td align="center">7</td>
<td align="center">6</td>
<td align="center">14</td>
<td align="center">7</td>
<td align="center">5</td>
<td align="center">9</td>
</tr>
<tr>
<td align="center">Number of regression trees/tree</td>
<td align="center">800</td>
<td align="center">950</td>
<td align="center">900</td>
<td align="center">1000</td>
<td align="center">850</td>
<td align="center">950</td>
<td align="center">850</td>
<td align="center">950</td>
<td align="center">550</td>
</tr>
<tr>
<td align="center">Test set coefficient of determinationR<sup>2</sup>
</td>
<td align="center">0.91</td>
<td align="center">0.90</td>
<td align="center">0.89</td>
<td align="center">0.89</td>
<td align="center">0.91</td>
<td align="center">0.89</td>
<td align="center">0.90</td>
<td align="center">0.89</td>
<td align="center">0.89</td>
</tr>
<tr>
<td align="center">Training set coefficient of determinationR<sup>2</sup>
</td>
<td align="center">0.93</td>
<td align="center">0.92</td>
<td align="center">0.93</td>
<td align="center">0.92</td>
<td align="center">0.92</td>
<td align="center">0.91</td>
<td align="center">0.94</td>
<td align="center">0.93</td>
<td align="center">0.92</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Comparative analysis of the prediction effect of three 9-class small-layer fracturing well capacity prediction regression models. It can be seen that the Random Forest regression algorithm established by the nine categories of fracturing well capacity prediction model has the smallest R<sup>2</sup> coefficient of determination, and the fracturing well capacity prediction GBRT regression model is better than the Bagging model as a whole.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 Case study</title>
<sec id="s3-1">
<title>3.1 Proppant volume optimization on the fracturing production rate</title>
<p>The proppant volume is one of the critical parameters influencing fractured well productivity. Designing an appropriate proppant volume prior to fracturing operations not only maximizes the oil-enhancement effects of stimulation but also effectively controls single-well operational costs and improves the cost-benefit ratio (<xref ref-type="bibr" rid="B3">Guo et al., 2024</xref>). Insufficient proppant volume leads to inadequate fracture width and uneven proppant distribution, reducing the effective stimulated reservoir volume and reservoir permeability. Conversely, excessive proppant volume may cause proppant flowback or poor packing, hindering the formation of high-conductivity fractures and even resulting in fracturing failure. Therefore, optimizing proppant volume is essential for ensuring fracturing efficacy and enhancing hydrocarbon productivity.</p>
<p>The impact of proppant volume on post-fracturing productivity can be analyzed using predictive models. Sensitivity analysis was conducted on proppant volume for three fracturing intervals from three wells using the productivity prediction model, as shown in <xref ref-type="fig" rid="F8">Figure 8</xref>. Key findings include that proppant volume exhibits distinct impact patterns on productivity across different intervals, yet each interval possesses an optimal proppant volume range. Within this range, fracturing achieves peak production enhancement for the specific interval.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Proppant volume optimization based on maximum production. <bold>(A)</bold> N1-1 <bold>(B)</bold> N1-2 <bold>(C)</bold> N1-3 <bold>(D)</bold> Z1-1 <bold>(E)</bold> Z1-2 <bold>(F)</bold> Z1-3 <bold>(G)</bold> B1-1 <bold>(H)</bold> B1-2 <bold>(I)</bold> B1-3.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g008.tif">
<alt-text content-type="machine-generated">A grouped image of nine charts shows the relationship between proppant volume in cubic meters and daily incremental oil production of fracturing intervals. Each chart indicates a specific peak point where the incremental oil production is highlighted. They are labeled (A) through (I), with each graph belonging to different sets identified as N1-1, N1-2, N1-3, Z1-1, Z1-2, Z1-3, B1-1, B1-2, and B1-3. Distinct colors and styles represent data points and trend lines for the different sets.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Capacity prediction after hydraulic fracturing</title>
<p>A total of 72 oil wells were randomly selected from out-of-sample datasets to validate the post-fracturing productivity GBRT model. The model predicted production rates for the first month after stimulation, which were then compared with actual field data. As illustrated in <xref ref-type="fig" rid="F9">Figure 9</xref>, the dashed lines represent actual daily incremental oil production, while the solid lines denote predicted values.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Predicted daily oil production vs. actual daily oil production.</p>
</caption>
<graphic xlink:href="feart-13-1601363-g009.tif">
<alt-text content-type="machine-generated">Line graph showing daily oil production for seventy-one wells. The red line with circles represents actual production, and blue dots depict projected production. Values range from zero to twelve tons. Trends fluctuate significantly across wells.</alt-text>
</graphic>
</fig>
<p>The performance metrics across the nine model categories in <xref ref-type="table" rid="T7">Table 7</xref>.</p>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Accuracy of prediction models in practical applications.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Model categories</th>
<th align="center">MRE</th>
<th align="center">RMSE</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">0</td>
<td align="center">7.94%</td>
<td align="center">0.26</td>
</tr>
<tr>
<td align="center">1</td>
<td align="center">11.33%</td>
<td align="center">0.35</td>
</tr>
<tr>
<td align="center">2</td>
<td align="center">17.12%</td>
<td align="center">0.42</td>
</tr>
<tr>
<td align="center">3</td>
<td align="center">14.66%</td>
<td align="center">0.45</td>
</tr>
<tr>
<td align="center">4</td>
<td align="center">10.97%</td>
<td align="center">0.44</td>
</tr>
<tr>
<td align="center">5</td>
<td align="center">9.92%</td>
<td align="center">0.57</td>
</tr>
<tr>
<td align="center">6</td>
<td align="center">12.32%</td>
<td align="center">0.58</td>
</tr>
<tr>
<td align="center">7</td>
<td align="center">13.18%</td>
<td align="center">0.60</td>
</tr>
<tr>
<td align="center">8</td>
<td align="center">9.47%</td>
<td align="center">0.85</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>MRE, ranges from7.94% (Category 0) to 17.12% (Category 2), with an average MRE, of 12.09%. RMSE, values are tightly clustered between 0.26 (Category 0) and 0.85 (Category 8), indicating robust model stability. The models demonstrate strong alignment with field observations, with 6 out of 9 categories achieving MRE &#x3c;13% and RMSE &#x3c;0.6, validating their predictive accuracy for post-fracturing performance.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>
<list list-type="simple">
<list-item>
<p>(1) Based on the physical mechanisms of horizontal fracture fracturing, relevant characteristic parameters were selected from geological conditions, production characteristics, and fracturing techniques to perform clustering analysis on fracturing intervals in the data sample. Similar intervals were categorized, and categorical modeling studies were conducted. This approach allows for matching the best model to different fracturing intervals according to their corresponding categories, thereby improving model applicability.</p>
</list-item>
<list-item>
<p>(2) For each category of fracturing intervals, correlation analysis was performed to identify the dominant controlling factors influencing post-fracturing productivity in each reservoir type. The dominant factors affecting post-fracturing performance differ slightly across interval categories. Machine learning algorithms such as Random forest, Bagging and GBRT were used to establish models describing the relationships between post-fracturing production enhancement effects, dominant factors, and production time for each reservoir category. The fracturing well capacity GBRT prediction model can predict productivity after fracturing in different intervals.</p>
</list-item>
<list-item>
<p>(3) In the era of smart oilfields, data-driven models hold broad application prospects for horizontal fracture fracturing productivity prediction. They can fully utilize data assets, uncover production patterns, compensate for the limitations of physical models, and improve the accuracy and reliability of post-fracturing productivity predictions.</p>
</list-item>
</list>
</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>QL: Methodology, Writing &#x2013; original draft. YS: Methodology, Writing &#x2013; review and editing, Formal Analysis. ML: Writing &#x2013; original draft, Data curation, Software. BG: Methodology, Writing &#x2013; review and editing. LL: Investigation, Conceptualization, Writing &#x2013; review and editing. YZ: Conceptualization, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>Author BG was employed by Engineering Technology Department of Daqing Oilfield Limited Company. Author LL was employed by Exploration and Development Research Institute of Dagang Oilfield Company. Author YZ was employed by Tianjin Branch of CNPC Logging.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The reviewer FZ declared a shared affiliation with the authors QL and YYS to the handling editor at the time of review.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/feart.2025.1601363/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/feart.2025.1601363/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Table2.xlsx" id="SM1" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table3.xlsx" id="SM2" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table9.xlsx" id="SM3" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table6.xlsx" id="SM4" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table4.xlsx" id="SM5" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table1.xlsx" id="SM6" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table5.xlsx" id="SM7" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table7.xlsx" id="SM8" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
<supplementary-material xlink:href="Table8.xlsx" id="SM9" mimetype="application/xlsx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>A systematic cluster analysis method for evaluating the effect of oil and gas wells after fracturing</article-title>. <source>Nat. Gas. Ind.</source> (<issue>10</issue>), <fpage>56</fpage>&#x2013;<lpage>58</lpage>. <pub-id pub-id-type="doi">10.3321/j.issn:1000-0976.2004.10.018</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Calculation of production increase multipliers for fractured wells</article-title>. <source>Oil Drill. Prod. Technol.</source> <volume>10</volume> (<issue>1</issue>), <fpage>67</fpage>&#x2013;<lpage>71</lpage>. <pub-id pub-id-type="doi">10.13639/j.odpt.1986.01.014</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Qu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Correlation analysis of construction yield of low-permeability fracturing with geological parameters and construction parameters in Sha II section of Bohai Sea region</article-title>. <source>China Petroleum Chem. Stand. Qual.</source> <volume>44</volume> (<issue>14</issue>), <fpage>154</fpage>&#x2013;<lpage>155&#x2b;158</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.1673-4076.2024.14.051</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A physics-informed neural network approach for surrogating a numerical simulation of fractured horizontal well production prediction</article-title>. <source>Energies</source> <volume>16</volume> (<issue>24</issue>), <fpage>7948</fpage>. <pub-id pub-id-type="doi">10.3390/en16247948</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). <source>A machine learning analysis based on big data for eagle ford shale formation</source>. <publisher-loc>Calgary, AB, Canada</publisher-loc>: <publisher-name>Society of Petroleum Engineers</publisher-name>, <fpage>1</fpage>&#x2013;<lpage>21</lpage>. <pub-id pub-id-type="doi">10.2118/196158-MS</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Horizontal well capacity prediction model for fractured horizontal wells in non-homogeneous tight reservoirs</article-title>. <source>Petroleum Geol. and Oilfield Dev. Daqing</source> <volume>41</volume> (<issue>04</issue>), <fpage>168</fpage>&#x2013;<lpage>174</lpage>. <pub-id pub-id-type="doi">10.19597/j.issn.1000-3754.202103021</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McGuire</surname>
<given-names>W. J.</given-names>
</name>
<name>
<surname>Sikora</surname>
<given-names>V. J.</given-names>
</name>
</person-group> (<year>1960</year>). <source>The effect of vertical fractures on well productivity</source>, <volume>1618</volume>.<source>SPE</source>. <pub-id pub-id-type="doi">10.2118/1618-G</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mohaghegh</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>GaskariA</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Maysami</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Shale analytics: making production and operational decisions based on facts: a case study in marcellus shale</article-title>. <source>SPE184822</source>. <pub-id pub-id-type="doi">10.2118/184822-MS</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Jing</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Volumetric fracturing capacity prediction study of horizontal wells in volcanic reservoirs</article-title>. <source>Lithol. Reserv.</source> <volume>30</volume> (<issue>03</issue>), <fpage>159</fpage>&#x2013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.12108/yxyqc.20180318</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prats</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>1961</year>). <source>Effect of vertical fractures on reservoir behavior-incompressible fluid case</source>, <volume>1575</volume>.<source>SPE</source>. <pub-id pub-id-type="doi">10.2118/1575-G</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raymond</surname>
<given-names>L. R.</given-names>
</name>
<name>
<surname>Binder</surname>
<given-names>G. G.</given-names>
<suffix>Jr.</suffix>
</name>
</person-group> (<year>1967</year>). <source>Productivity of wells in vertically fractured managed formation</source>, <volume>1454</volume>.<source>SPE</source>. <pub-id pub-id-type="doi">10.2118/1454-PA</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>XGBoost-based capacity prediction for fractured horizontal wells</article-title>. <source>China Petroleum Chem. Stand. Qual.</source> <volume>43</volume> (<issue>24</issue>), <fpage>15</fpage>&#x2013;<lpage>17</lpage>. <pub-id pub-id-type="doi">10.3969/j.issn.1673-4076.2023.24.006</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Insights to fracture stimulation design in unconventional reservoirs based on machine learning modeling</article-title>. <source>J. Petroleum Sci. Eng.</source> <volume>174</volume>, <fpage>682</fpage>&#x2013;<lpage>695</lpage>. <pub-id pub-id-type="doi">10.1016/j.petrol.2018.11.076</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Application of cluster analysis in selection of wells and intervals for repeated fracturing of horizontal wells</article-title>. <source>Chem. Eng. Equip.</source> (<issue>03</issue>), <fpage>76</fpage>&#x2013;<lpage>78</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Data-driven post-fracturing production prediction based on tight sandstones in Changqing oilfield</article-title>. <source>China Energy Environ. Prot.</source> <volume>43</volume> (<issue>10</issue>), <fpage>96</fpage>&#x2013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.19389/j.cnki.1003-0506.2021.10.018</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Research on fractured well fluid production prediction technology based on production increase multiplier</article-title>. <source>Sino-Global Energy</source> <volume>25</volume> (<issue>04</issue>), <fpage>50</fpage>&#x2013;<lpage>56</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>