<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" article-type="research-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Public Health</journal-id>
<journal-title-group>
<journal-title>Frontiers in Public Health</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Public Health</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2296-2565</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpubh.2025.1616841</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>The health cost of urbanization: identification and ranking of influencing factors of class A and B infectious diseases based on machine learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Zhiqing</given-names>
</name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2874293"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhu</surname>
<given-names>Boyi</given-names>
</name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3080040"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Liuyu</given-names>
</name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/3288818"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x0026; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x0026; editing</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>School of Public Policy and Management, Tsinghua University</institution>, <city>Beijing</city>, <country country="cn">China</country></aff>
<aff id="aff2"><label>2</label><institution>Southwest Jiaotong University</institution>, <city>Chengdu</city>, <country country="cn">China</country></aff>
<aff id="aff3"><label>3</label><institution>School of Accounting, Southwestern University of Finance and Economics</institution>, <city>Chengdu</city>, <country country="cn">China</country></aff>
<author-notes>
<corresp id="c001"><label>&#x002A;</label>Correspondence: Boyi Zhu, <email xlink:href="mailto:annyzby@163.com">annyzby@163.com</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-12-03">
<day>03</day>
<month>12</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>13</volume>
<elocation-id>1616841</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>09</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>11</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Wang, Zhu and Wu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wang, Zhu and Wu</copyright-holder>
<license>
<ali:license_ref start_date="2025-12-03">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Aims</title>
<p>With the rapid advancement of urbanization, an increasing number of people are congregating in urban areas, leading to a higher density of economic activities. This may not only accelerate the spread of infectious diseases but also result in pollutants that harm residents&#x2019; health. Nevertheless, improvements in infrastructure and healthcare services, coupled with heightened awareness of personal protection among residents, can effectively mitigate the spread of infectious diseases. The incidence rates of Class A and B infectious diseases serve as critical indicators of public health status. This study seeks to identify and prioritize the key factors influencing public health during the process of rapid urbanization, thereby providing a scientific basis for decision-making aimed at enhancing residents&#x2019; living environments and addressing existing gaps in public health systems.</p>
</sec>
<sec>
<title>Methods</title>
<p>Using provincial-level data from mainland China (2008&#x2013;2022), this study systematically applied multiple machine learning methods, including Random Forest, Gradient Boosting Tree, and XGBoost, to evaluate the impacts of over 10 indicators across economic, demographic, land use, and social dimensions on the Incidence Rate of Class A and B Infectious Diseases per 100,000 Population.</p>
</sec>
<sec>
<title>Results</title>
<p>(1) Social factors account for 46.7%, constituting the most significant determinant, succeeded by land use, population, and economic dimensions. (2) Public transportation, urban water supply coverage, healthcare expenditure, and the spatial distribution of healthcare resources exert direct effects on residents&#x2019; health outcomes and the accessibility of public health services. (3) Regarding land use, effective urban planning&#x2014;reflected in indicators such as the green coverage rate of built-up areas and the per capita area of paved roads&#x2014;plays a crucial role in promoting public health, accounting for 17.4% on average, whereas inadequate land-use management often precipitates health risks. (4) Population dynamics, encompassing demographic restructuring, agglomeration, and education levels, simultaneously generate advantages (e.g., improved efficiency in health service delivery) and challenges (e.g., heightened vulnerability to infectious disease transmission). (5) Economic factors, including industrial pollution control (ratio of completed investment in industrial pollution control to GDP), industrial upgrading (ratio of tertiary to secondary industry value-added), international trade (foreign trade dependence), and income levels (per capita disposable income of urban residents), manifest dual effects: advancing health improvements while engendering environmental degradation and cross-border health risks. (6) The health implications of rapid urbanization display regional disparity: social factors (60.2%) predominate in eastern China, economic factors (63.2%) in central China, and land use factors (54.2%) in western regions.</p>
</sec>
<sec>
<title>Conclusion</title>
<p>During rapid urbanization, governments must prioritize timely enhancements to public health services, rational land use planning, and protection of vulnerable populations. Emphasizing the quality of economic development and fostering synergies between industrial upgrading and environmental governance will improve public health outcomes.</p>
</sec>
</abstract>
<kwd-group>
<kwd>public health</kwd>
<kwd>infectious diseases</kwd>
<kwd>urbanization</kwd>
<kwd>machine learning</kwd>
<kwd>influencing factors</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This research is funded by the Fundamental Research Funds for the Central Universities (2682025CX087, 2682025CX119), Yibin City Science and Technology program (2024MZ001).</funding-statement>
</funding-group>
<counts>
<fig-count count="3"/>
<table-count count="9"/>
<equation-count count="0"/>
<ref-count count="40"/>
<page-count count="13"/>
<word-count count="8758"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Environmental Health and Exposome</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>With the rapid acceleration of global urbanization, human settlement patterns are experiencing an unprecedented structural transformation. According to statistics, approximately 55% of the global population currently resides in urban areas, and this proportion is projected to increase to 68% by 2050 (<xref ref-type="bibr" rid="ref1">1</xref>). This process not only reshapes urban spatial configurations but also profoundly transforms residents&#x2019; living environments. On one hand, it leads to significant improvements in material living conditions, such as optimized dietary structures, enhanced housing quality, and improved transportation networks. On the other hand, it has a substantial impact on residents&#x2019; physical and mental health (<xref ref-type="bibr" rid="ref1 ref2 ref3">1&#x2013;3</xref>). Public health levels are crucial not only for individual development and family well-being but also for influencing regional economic efficiency and social governance through the mechanism of human capital accumulation (<xref ref-type="bibr" rid="ref4">4</xref>, <xref ref-type="bibr" rid="ref5">5</xref>). However, urbanization and public health are not inherently synchronized. According to World Bank data, in the 21st century, the growth rate of the urban population in middle-income and low-income countries has significantly exceeded the global average. Despite this rapid urbanization, these regions have also experienced higher incidences of infectious diseases, revealing a structural imbalance between economic development and social governance. To effectively curb the spread of infectious diseases, enhance the public health system, and harmonize the relationship between economic development and public health, it is crucial to systematically analyze the determinants of public health and establish a multi-dimensional indicator framework that integrates economic development, demographic changes, land use patterns, and social infrastructure. This framework will provide a robust scientific foundation for decision-making aimed at creating healthy and livable cities, ultimately fostering a dynamic equilibrium between urban and individual development.</p>
<p>Urbanization exerts a dual and complex influence on public health. On the one hand, it reflects not only the rising proportion of urban populations but also advancements in infrastructure and the agglomeration of high-end resources, all of which contribute positively to improving public health conditions. On the other hand, the urbanization process is often accompanied by accelerated industrialization, transformations in residential and lifestyle patterns, and increased population density&#x2014;factors that can amplify the spread of infectious diseases and exacerbate environmental pollution. These developments, in turn, pose significant threats to population health (<xref ref-type="bibr" rid="ref2">2</xref>, <xref ref-type="bibr" rid="ref6 ref7 ref8 ref9 ref10">6&#x2013;10</xref>).</p>
<p>Particularly during periods of rapid urbanization, insufficient or lagging infrastructure development and uncoordinated urban planning can lead to an uneven distribution of healthcare resources and facilitate the accelerated transmission of infectious diseases (<xref ref-type="bibr" rid="ref11 ref12 ref13 ref14 ref15 ref16">11&#x2013;16</xref>). However, existing studies on the determinants of public health often concentrate on single or narrowly defined indicators (<xref ref-type="bibr" rid="ref17">17</xref>, <xref ref-type="bibr" rid="ref18">18</xref>), limiting their capacity to holistically assess the broad and multifaceted impacts of urbanization on health outcomes.</p>
<p>Moreover, conventional research methodologies frequently exhibit significant limitations in capturing the full scope of urbanization-related variables (<xref ref-type="bibr" rid="ref19 ref20 ref21 ref22">19&#x2013;22</xref>), with a restricted range of influencing factors considered. This methodological constraint hinders a comprehensive understanding of the diversity and complexity inherent in the urbanization&#x2013;health nexus. In reality, the relationship between urbanization and public health is highly dynamic and context-dependent, with dominant influencing factors varying significantly across different stages of urban development (<xref ref-type="bibr" rid="ref23">23</xref>).</p>
<p>In contrast to traditional statistical methods, machine learning does not presuppose a functional form, thereby circumventing issues of functional misspecification (<xref ref-type="bibr" rid="ref24">24</xref>). For instance, while public health is closely related to economic development levels and demographic factors, it is impractical to predefine an explicit functional relationship. Leveraging nonparametric estimation techniques, machine learning offers greater flexibility by identifying and retaining the most salient features from a vast array of variables without manual selection, thus partially mitigating the curse of dimensionality (<xref ref-type="bibr" rid="ref25">25</xref>). A case in point is the multifactorial nature of public health, where pinpointing dominant influencing factors poses both a critical and pragmatic challenge. Furthermore, machine learning exhibits robust generalization capabilities, making it well-suited for predictive scenarios (<xref ref-type="bibr" rid="ref26">26</xref>). These methods mitigate overfitting through cross-validation techniques and balance the trade-off between &#x201C;predictive performance&#x201D; and &#x201C;generalization ability&#x201D; through regularization strategies, as seen in algorithms such as Random Forest. Machine learning models like Random Forest require minimal hyperparameter tuning and are less dependent on prior assumptions about the data (<xref ref-type="bibr" rid="ref27">27</xref>). To address the limitations of existing research, this study leverages provincial panel data from mainland China spanning 2008 to 2022. It constructs a system of over 10 indicators across four key dimensions&#x2014;economic development, population change, land use patterns, and social infrastructure&#x2014;and applies machine learning algorithms, including Random Forest, Gradient Boosting Trees (GDBT), and XGBoost, to quantitatively analyze the relative contributions of these dimensions to the incidence rates of class A and B statutory infectious diseases. This approach not only uncovers the complex, multifactorial causal mechanisms underlying the incidence rates of these infectious diseases during the urbanization process but also provides an evidence-based foundation for optimizing disease prevention and control policies. What&#x2019;s more, Random Forest, GDBT, and XGBoost are robust (<xref ref-type="bibr" rid="ref28">28</xref>), are adopted to represent both Bagging and Boosting families in ensemble learning. These three models have demonstrated excellent stability, interpretability, and generalization performance in empirical studies involving limited-sample socioeconomic and health data (<xref ref-type="bibr" rid="ref27">27</xref>, <xref ref-type="bibr" rid="ref29">29</xref>). These findings offer valuable insights for both public health theory and practice, highlighting critical factors influencing disease dynamics in the context of urban development. <xref ref-type="fig" rid="fig1">Figure 1</xref> shows the research dimensions and research methods.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Research dimensions and research methods.</p>
</caption>
<graphic xlink:href="fpubh-13-1616841-g001.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Diagram showing the component of urbanization involving society, economy, population, and land, connected by arrows. Next, machine learning models like Random Forest, Gradient Boosted Tree, and XGBoost lead to public health outcomes.</alt-text>
</graphic>
</fig>
</sec>
<sec sec-type="materials|methods" id="sec2">
<label>2</label>
<title>Materials and methods</title>
<sec id="sec3">
<label>2.1</label>
<title>Data source</title>
<p>This study selects data from 30 provincial-level units on the mainland of China, covering the period from 2008 to 2022. Hong Kong, Macau, Taiwan, and Tibet are excluded from the research scope due to significant data gaps. Based on relevant literature (<xref ref-type="bibr" rid="ref20">20</xref>, <xref ref-type="bibr" rid="ref40">40</xref>), the input variables are categorized into 16 variables in four major factors, namely, demographic, economic, land and social factors, and the data sources are all from the National Bureau of Statistics as well as provincial statistical yearbooks, while the output variable is the incidence rate of class A and B infectious diseases per 100,000 population, and the data source is the China Health Statistical Yearbook. The Total Dependency Ratio is not published for 2020. According to its definition, this study imputes the 2020 value based on census data, using the ratio of the non-working-age population (0&#x2013;14 and 65+) to the working-age population (15&#x2013;64). See <xref ref-type="table" rid="tab1">Table 1</xref> for descriptive statistics on the classification of variable categories.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Descriptive statistical characteristics of variables.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Type</th>
<th align="left" valign="top">Variables</th>
<th align="center" valign="top"><italic>N</italic></th>
<th align="center" valign="top">Mean</th>
<th align="center" valign="top">Std</th>
<th align="center" valign="top">Min</th>
<th align="center" valign="top">Max</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Output variable</td>
<td align="left" valign="middle">Incidence rate of class A and B infectious diseases per 100,000 population</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">240.13</td>
<td align="char" valign="middle" char=".">99.63</td>
<td align="char" valign="middle" char=".">74.39</td>
<td align="char" valign="middle" char=".">738.19</td>
</tr>
<tr>
<td align="left" valign="middle">Dem1</td>
<td align="left" valign="middle">Total dependency ratio</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">38.08</td>
<td align="char" valign="middle" char=".">7.453</td>
<td align="char" valign="middle" char=".">19.30</td>
<td align="char" valign="middle" char=".">57.79</td>
</tr>
<tr>
<td align="left" valign="middle">Dem2</td>
<td align="left" valign="middle">Urbanization rate</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">57.71</td>
<td align="char" valign="middle" char=".">12.994</td>
<td align="char" valign="middle" char=".">29.11</td>
<td align="char" valign="middle" char=".">89.60</td>
</tr>
<tr>
<td align="left" valign="middle">Dem3</td>
<td align="left" valign="middle">Illiteracy rate</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">5.10</td>
<td align="char" valign="middle" char=".">3.066</td>
<td align="char" valign="middle" char=".">0.79</td>
<td align="char" valign="middle" char=".">17.78</td>
</tr>
<tr>
<td align="left" valign="middle">Dem4</td>
<td align="left" valign="middle">Urban population density</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">2889.06</td>
<td align="char" valign="middle" char=".">1169.59</td>
<td align="char" valign="middle" char=".">649.00</td>
<td align="char" valign="middle" char=".">5967.00</td>
</tr>
<tr>
<td align="left" valign="middle">Eco1</td>
<td align="left" valign="middle">Ratio of tertiary industry value-added to secondary industry value-added</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">1.29</td>
<td align="char" valign="middle" char=".">0.722</td>
<td align="char" valign="middle" char=".">0.53</td>
<td align="char" valign="middle" char=".">5.24</td>
</tr>
<tr>
<td align="left" valign="middle">Eco2</td>
<td align="left" valign="middle">Foreign trade dependence</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">0.28</td>
<td align="char" valign="middle" char=".">0.307</td>
<td align="char" valign="middle" char=".">0.01</td>
<td align="char" valign="middle" char=".">1.60</td>
</tr>
<tr>
<td align="left" valign="middle">Eco3</td>
<td align="left" valign="middle"><italic>Per Capita</italic> disposable income of urban residents</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">30504.56</td>
<td align="char" valign="middle" char=".">13393.60</td>
<td align="char" valign="middle" char=".">11413.00</td>
<td align="char" valign="middle" char=".">84034.00</td>
</tr>
<tr>
<td align="left" valign="middle">Eco4</td>
<td align="left" valign="middle">Ratio of completed investment in industrial pollution control to GDP</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">0.12</td>
<td align="char" valign="middle" char=".">0.125</td>
<td align="char" valign="middle" char=".">0.00</td>
<td align="char" valign="middle" char=".">1.10</td>
</tr>
<tr>
<td align="left" valign="middle">Land1</td>
<td align="left" valign="middle">Ratio of built-up area to urban administrative area</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">32.53</td>
<td align="char" valign="middle" char=".">14.253</td>
<td align="char" valign="middle" char=".">7.50</td>
<td align="char" valign="middle" char=".">70.98</td>
</tr>
<tr>
<td align="left" valign="middle">Land2</td>
<td align="left" valign="middle">Green coverage rate of urban built-up areas</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">39.35</td>
<td align="char" valign="middle" char=".">3.990</td>
<td align="char" valign="middle" char=".">25.90</td>
<td align="char" valign="middle" char=".">49.80</td>
</tr>
<tr>
<td align="left" valign="middle">Land3</td>
<td align="left" valign="middle"><italic>Per Capita</italic> area of paved roads</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">15.71</td>
<td align="char" valign="middle" char=".">5.084</td>
<td align="char" valign="middle" char=".">4.04</td>
<td align="char" valign="middle" char=".">28.00</td>
</tr>
<tr>
<td align="left" valign="middle">Soc1</td>
<td align="left" valign="middle">Number of public transport vehicles in standard units per 10,000 population</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">12.34</td>
<td align="char" valign="middle" char=".">3.125</td>
<td align="char" valign="middle" char=".">6.83</td>
<td align="char" valign="middle" char=".">26.55</td>
</tr>
<tr>
<td align="left" valign="middle">Soc2</td>
<td align="left" valign="middle">Urban water supply coverage rate</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">97.42</td>
<td align="char" valign="middle" char=".">3.100</td>
<td align="char" valign="middle" char=".">82.03</td>
<td align="char" valign="middle" char=".">100.00</td>
</tr>
<tr>
<td align="left" valign="middle">Soc3</td>
<td align="left" valign="middle">Proportion of government health expenditure in total fiscal expenditure</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">7.61</td>
<td align="char" valign="middle" char=".">1.637</td>
<td align="char" valign="middle" char=".">3.90</td>
<td align="char" valign="middle" char=".">13.93</td>
</tr>
<tr>
<td align="left" valign="middle">Soc4</td>
<td align="left" valign="middle">Number of urban assistant practicing physicians per 10,000 population</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">40.84</td>
<td align="char" valign="middle" char=".">7.345</td>
<td align="char" valign="middle" char=".">22.29</td>
<td align="char" valign="middle" char=".">60.01</td>
</tr>
<tr>
<td align="left" valign="middle">Soc5</td>
<td align="left" valign="middle">Number of healthcare institution beds per 10,000 population</td>
<td align="center" valign="middle">450</td>
<td align="char" valign="middle" char=".">51.33</td>
<td align="char" valign="middle" char=".">14.059</td>
<td align="char" valign="middle" char=".">23.11</td>
<td align="char" valign="middle" char=".">84.32</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Random forest</title>
<p>The Random Forest algorithm, proposed by Breiman (<xref ref-type="bibr" rid="ref30">30</xref>), is an ensemble learning-based method that extends the Bagging algorithm. It trains a decision tree on each subset and randomly selects a subset of features for splitting at each node, increasing feature randomness and reducing correlations between decision trees. The prediction results are obtained by averaging or voting the predictions of all trees. This process helps to reduce variance and errors, preventing both overfitting and underfitting. Each feature is ranked according to its assigned weight, referred to as feature importance (<xref ref-type="bibr" rid="ref31">31</xref>). Through multiple random sampling and feature selection, Random Forest effectively handles high-dimensional data and uses tree-based ensemble methods to capture complex non-linear relationships between variables. Random forests evaluate feature contributions by calculating the Gini index after each feature split. Feature importance is defined as the total reduction in Gini impurity achieved by all splits involving that feature across the ensemble of trees. A higher value indicates a greater contribution of the feature to the classification task.</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>Gradient Boosting Decision Tree (GBDT)</title>
<p>Gradient Boosting Decision Tree (GBDT) is a Boosting algorithm with CART regression trees as the base learners, primarily aimed at optimizing general loss functions. The core idea is to fit the residuals of the previous round&#x2019;s base learner using the negative gradient of the loss function, progressively reducing the residuals with each round. This approach ensures that the output of each round&#x2019;s base learner approaches the true values. By fitting along the negative gradient direction, the method guarantees that the loss function decreases as rapidly as possible in each training round, accelerating convergence to a local or global optimal solution (<xref ref-type="bibr" rid="ref29">29</xref>). GBDT is a powerful ensemble learning method capable of handling both continuous and discrete values. By iteratively optimizing the loss function, it captures complex nonlinear relationships, and the stage-by-stage residual fitting process ensures robustness even in the presence of noise, with regularization parameters effectively preventing overfitting. The calculation of feature importance in GBDT essentially quantifies &#x201C;the aggregate contribution of a feature to the reduction of the loss function across all trees.&#x201D; The more frequently a feature is utilized for node splitting, and the more significantly each split reduces the loss function, the higher its importance weight.</p>
</sec>
<sec id="sec6">
<label>2.4</label>
<title>XGBoost</title>
<p>XGBoost is an efficient implementation of GBDT, which first uses the training set and sample true values to train a tree. This tree is then used to predict the training set, and the residuals (i.e., the differences between the predicted and true values) are calculated. The residuals are then used as the target for training the second tree. After the second tree is trained, the residuals are recalculated for each sample, and a third tree is trained, and so on (<xref ref-type="bibr" rid="ref32">32</xref>). XGBoost utilizes a second-order Taylor expansion to more accurately approximate the optimal solution of the loss function, thereby improving prediction accuracy. During decision tree construction, it uses an &#x201C;approximate greedy algorithm&#x201D; to reduce computational complexity. The method also supports custom loss functions and evaluation metrics, allowing model parameters to be adjusted according to the research problem. Additionally, it can handle missing values automatically and supports parallel computation. XGBoost quantifies feature importance by measuring the reduction in the objective function (e.g., loss function) achieved at each split, referred to as the &#x201C;gain.&#x201D; After each split, the gain score of feature A accumulates the reduction in loss attributable to that split. The final importance score for feature A is obtained by summing its total gains across all trees. A higher gain indicates that the feature plays a more significant role in distinguishing samples.</p>
</sec>
<sec id="sec7">
<label>2.5</label>
<title>Performance evaluation</title>
<p>The Root Mean Squared Error (RMSE), Mean Absolute Percentage Error (MAPE), and Mean Absolute Error (MAE) are the main metrics for evaluating the model&#x2019;s prediction performance. The coefficient of determination (<italic>R</italic><sup>2</sup>) is used to assess the goodness of fit of the model, and a higher <italic>R</italic><sup>2</sup> value indicates a superior goodness of fit. MAE represents the average absolute error between predicted and actual values, MAPE indicates the degree of data dispersion, and RMSE measures both the error between actual and predicted values and the dispersion between the two errors. The smaller these metrics, the better the model&#x2019;s performance. However, all three metrics are influenced by sample values and predicted values, and there is no predefined standard. Given that K-fold cross-validation is appropriate for small datasets and facilitates effective use of all data for training and evaluation, and to ensure that each fold contains an adequate number of samples during cross-validation while preventing sample structure imbalance, this study adopts three-fold and four-fold cross-validation. The evaluation is conducted using holdout samples under various training/testing splits (70, 80, and 90%). For Boosting-based models (GBDT, XGBoost), their mechanism of sequentially fitting residuals tends to drive training error toward 0 and training <italic>R</italic><sup>2</sup> toward 1 when the number of iterations increases. Since RMSE has a broader evaluation range, this study chooses <italic>R</italic><sup>2</sup> and RMSE to evaluate prediction performance.</p>
</sec>
<sec id="sec8">
<label>2.6</label>
<title>Software implementation</title>
<p>In this study, SPSSPRO is used to execute all algorithms, employing machine learning regression modules such as Random Forest, GBDT, and XGBoost. <xref ref-type="table" rid="tab1">Table 1</xref> shows the descriptive statistical characteristics of variables.</p>
</sec>
</sec>
<sec sec-type="results" id="sec9">
<label>3</label>
<title>Results</title>
<p>Prior to modeling, a correlation analysis is conducted across all variables. While the correlation coefficient between &#x2018;Ratio of Built-up Area to Urban Administrative Area&#x2019; and &#x2018;Urban Population Density&#x2019; exceeds 0.8, all other inter-variable correlations remain below 0.7. Given that these two variables capture distinct aspects&#x2014;the role of land use and population dynamics in urbanization&#x2014;they are retained in the model. All machine learning methods employed in this study incorporated data shuffling and cross-validation procedures. Using random forest as the baseline model, parameter tuning and methodological adjustments are systematically implemented to identify robust computational outcomes.</p>
<sec id="sec10">
<label>3.1</label>
<title>Random forest results</title>
<p>Initial implementation allocates 70% of samples to the training set and 30% to the test set, with three-fold cross-validation. As <xref ref-type="table" rid="tab2">Table 2</xref> shows, the results indicate a training set RMSE of 27.454 and a test set RMSE of 48.524, corresponding to coefficients of determination (<italic>R</italic><sup>2</sup>) of 0.93 and 0.696, respectively.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Comparison of predictive power for different training set shares.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th rowspan="2">Set</th>
<th align="center" valign="top" colspan="2">70% training set share</th>
<th align="center" valign="top" colspan="2">80% training set share</th>
<th align="center" valign="top" colspan="2">90% training set share</th>
</tr>
<tr>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Training set</td>
<td align="char" valign="middle" char=".">27.454</td>
<td align="char" valign="middle" char=".">0.93</td>
<td align="char" valign="middle" char=".">24.796</td>
<td align="char" valign="middle" char=".">0.937</td>
<td align="char" valign="middle" char=".">25.113</td>
<td align="char" valign="middle" char=".">0.938</td>
</tr>
<tr>
<td align="left" valign="middle">Test set</td>
<td align="char" valign="middle" char=".">48.542</td>
<td align="char" valign="middle" char=".">0.696</td>
<td align="char" valign="middle" char=".">54.869</td>
<td align="char" valign="middle" char=".">0.708</td>
<td align="char" valign="middle" char=".">41.859</td>
<td align="char" valign="middle" char=".">0.773</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Subsequent trials adjusted the training set proportions to 80 and 90%. While test set RMSE decreased with larger training sets, the 80% training allocation exhibits lower training set RMSE (23.117) compared to the 90% partition (25.832). However, the 90% training configuration demonstrates superior generalization performance, achieving a higher <italic>R</italic><sup>2</sup>. Comparative analysis of these metrics suggests the 90% training ratio optimizes predictive accuracy and model fit.</p>
<p>To mitigate potential overfitting risks inherent in high training set allocations (90%), this study adopts complementary methodologies to enhance model generalizability and investigates the model&#x2019;s adaptability to varying data volumes. Subsequent sections detail these refinements and their validation outcomes.</p>
</sec>
<sec id="sec11">
<label>3.2</label>
<title>Comparison of three algorithms</title>
<p><xref ref-type="table" rid="tab3">Table 3</xref> presents the performance metrics of three algorithms under varying training ratios and cross-validation folds.</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Comparison of predictive ability of different algorithms.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th rowspan="2">Set and Scenario</th>
<th align="center" valign="top" colspan="2">Random Forest</th>
<th align="center" valign="top" colspan="2">GBDT</th>
<th align="center" valign="top" colspan="2">XGBoost</th>
</tr>
<tr>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle" colspan="7">Scenario 1: 70% training set share, 3 folds</td>
</tr>
<tr>
<td align="left" valign="middle">Training set</td>
<td align="char" valign="middle" char=".">27.454</td>
<td align="char" valign="middle" char=".">0.93</td>
<td align="char" valign="middle" char=".">0.124</td>
<td align="center" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.573</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="left" valign="middle">Test set</td>
<td align="char" valign="middle" char=".">48.542</td>
<td align="char" valign="middle" char=".">0.696</td>
<td align="char" valign="middle" char=".">47.721</td>
<td align="center" valign="middle">0.754</td>
<td align="char" valign="middle" char=".">51.397</td>
<td align="center" valign="middle">0.735</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="7">Scenario 2: 80% training set share, 3 folds</td>
</tr>
<tr>
<td align="left" valign="middle">Training Set</td>
<td align="char" valign="middle" char=".">24.796</td>
<td align="char" valign="middle" char=".">0.937</td>
<td align="char" valign="middle" char=".">0.216</td>
<td align="center" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.406</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="left" valign="middle">Test set</td>
<td align="char" valign="middle" char=".">54.869</td>
<td align="char" valign="middle" char=".">0.708</td>
<td align="char" valign="middle" char=".">44.278</td>
<td align="center" valign="middle">0.78</td>
<td align="char" valign="middle" char=".">62.001</td>
<td align="center" valign="middle">0.371</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="7">Scenario 3: 70% training set share, 4 folds</td>
</tr>
<tr>
<td align="left" valign="middle">Training set</td>
<td align="char" valign="middle" char=".">24.402</td>
<td align="char" valign="middle" char=".">0.94</td>
<td align="char" valign="middle" char=".">0.224</td>
<td align="center" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.386</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="left" valign="middle">Test set</td>
<td align="char" valign="middle" char=".">52.471</td>
<td align="char" valign="middle" char=".">0.718</td>
<td align="char" valign="middle" char=".">45.725</td>
<td align="center" valign="middle">0.778</td>
<td align="char" valign="middle" char=".">45.741</td>
<td align="center" valign="middle">0.774</td>
</tr>
<tr>
<td align="left" valign="middle" colspan="7">Scenario 4: 80% training set share, 4 folds</td>
</tr>
<tr>
<td align="left" valign="middle">Training set</td>
<td align="char" valign="middle" char=".">24.742</td>
<td align="char" valign="middle" char=".">0.941</td>
<td align="char" valign="middle" char=".">0.252</td>
<td align="center" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.543</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="left" valign="middle">Test set</td>
<td align="char" valign="middle" char=".">52.564</td>
<td align="char" valign="middle" char=".">0.662</td>
<td align="char" valign="middle" char=".">45.734</td>
<td align="center" valign="middle">0.829</td>
<td align="char" valign="middle" char=".">36.48</td>
<td align="center" valign="middle">0.772</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In Scenario 1, the GBDT algorithm achieved the lowest RMSE values (training set: 0.124; test set: 47.721), significantly outperforming the other two algorithms. Correspondingly, its coefficients of determination (<italic>R</italic><sup>2</sup>) for both training and test sets surpasses those of alternative methods. Similar trends are observed in Scenario 2. However, in Scenario 3, GBDT and XGBoost demonstrated comparable performance without statistically significant differences in metrics. In Scenario 4, GBDT exhibits a lower test set RMSE than XGBoost, while achieving the highest <italic>R</italic><sup>2</sup> value.</p>
<p>Six optimal configurations are subsequently identified: GBDT (Scenario 1), GBDT (Scenario 2), GBDT/XGBoost (Scenario 3), and GBDT/XGBoost (Scenario 4). Comparative analysis reveals that GBDT (Scenario 3) delivers the lowest RMSE values across both training and test sets. Although its <italic>R</italic><sup>2</sup> (0.783) marginally trails that of GBDT (Scenario 4, <italic>R</italic><sup>2</sup>&#x202F;=&#x202F;0.812), the negligible performance gap (&#x003C;3%) prioritizes RMSE minimization for model selection. Consequently, GBDT (Scenario 3) is deemed the optimal predictor, followed by GBDT (Scenario 4) and XGBoost (Scenario 3). The boosting models achieve near-perfect fits on the training data (<italic>R</italic><sup>2</sup>&#x202F;=&#x202F;1) while exhibiting a generalization gap on the test sets&#x2014;a common pattern under noisy observational data. The selected specification best balances bias and variance, yielding the lowest test RMSE. Subsequent analyses focuses on these three configurations.</p>
</sec>
<sec id="sec12">
<label>3.3</label>
<title>Variable importance ranking</title>
<p><xref ref-type="table" rid="tab4">Table 4</xref> summarizes variable importance rankings across three selected configurations. Social factors consistently dominated all models, accounting for over 45% of total feature weights, with GBDT (Scenario 4) reaching 0.477.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>Ranking of the importance of each type of factor.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Factor</th>
<th align="center" valign="top">GBDT<break/>(Scenario 3)</th>
<th align="center" valign="top">GBDT<break/>(Scenario 4)</th>
<th align="center" valign="top">XGBoost<break/>(Scenario 3)</th>
<th align="center" valign="top">Average results</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Society</td>
<td align="char" valign="middle" char=".">0.457</td>
<td align="char" valign="middle" char=".">0.477</td>
<td align="char" valign="middle" char=".">0.467</td>
<td align="char" valign="middle" char=".">0.467</td>
</tr>
<tr>
<td align="left" valign="middle">Land</td>
<td align="char" valign="middle" char=".">0.115</td>
<td align="char" valign="middle" char=".">0.229</td>
<td align="char" valign="middle" char=".">0.301</td>
<td align="char" valign="middle" char=".">0.215</td>
</tr>
<tr>
<td align="left" valign="middle">Population</td>
<td align="char" valign="middle" char=".">0.282</td>
<td align="char" valign="middle" char=".">0.156</td>
<td align="char" valign="middle" char=".">0.074</td>
<td align="char" valign="middle" char=".">0.17</td>
</tr>
<tr>
<td align="left" valign="middle">Economy</td>
<td align="char" valign="middle" char=".">0.147</td>
<td align="char" valign="middle" char=".">0.139</td>
<td align="char" valign="middle" char=".">0.158</td>
<td align="char" valign="middle" char=".">0.148</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Discrepancies emerged among secondary factors: GBDT (Scenario 3) prioritized demographic factors (28.1%), followed by economic (15.6%) and land-related (11.3%) elements, whereas GBDT (Scenario 4) and XGBoost (Scenario 3) assigned higher weights to land factors (22.4 and 19.8%, respectively), relegating demographic and economic factors to tertiary positions. A composite ranking derived from mean importance scores established the hierarchy as: social factors &#x003E; land factors &#x003E; demographic factors &#x003E; economic factors. Quantitative weight calculations are shown in <xref ref-type="table" rid="tab5">Table 5</xref> and <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>Results of variable weight calculation.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Type</th>
<th align="left" valign="top">Variables</th>
<th align="center" valign="top">GBDT<break/>(Scenario 3)</th>
<th align="center" valign="top">GBDT<break/>(Scenario 4)</th>
<th align="center" valign="top">XGBoost<break/>(Scenario 3)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Dem1</td>
<td align="left" valign="middle">Total dependency ratio</td>
<td align="char" valign="middle" char=".">0.088</td>
<td align="char" valign="middle" char=".">0.065</td>
<td align="char" valign="middle" char=".">0.024</td>
</tr>
<tr>
<td align="left" valign="middle">Dem2</td>
<td align="left" valign="middle">Urbanization rate</td>
<td align="char" valign="middle" char=".">0.136</td>
<td align="char" valign="middle" char=".">0.016</td>
<td align="char" valign="middle" char=".">0.012</td>
</tr>
<tr>
<td align="left" valign="middle">Dem3</td>
<td align="left" valign="middle">Illiteracy rate</td>
<td align="char" valign="middle" char=".">0.042</td>
<td align="char" valign="middle" char=".">0.053</td>
<td align="char" valign="middle" char=".">0.024</td>
</tr>
<tr>
<td align="left" valign="middle">Dem4</td>
<td align="left" valign="middle">Urban population density</td>
<td align="char" valign="middle" char=".">0.016</td>
<td align="char" valign="middle" char=".">0.022</td>
<td align="char" valign="middle" char=".">0.014</td>
</tr>
<tr>
<td align="left" valign="middle">Eco1</td>
<td align="left" valign="middle">Ratio of tertiary industry value-added to secondary industry value-added</td>
<td align="char" valign="middle" char=".">0.04</td>
<td align="char" valign="middle" char=".">0.035</td>
<td align="char" valign="middle" char=".">0.084</td>
</tr>
<tr>
<td align="left" valign="middle">Eco2</td>
<td align="left" valign="middle">Foreign trade dependence</td>
<td align="char" valign="middle" char=".">0.09</td>
<td align="char" valign="middle" char=".">0.071</td>
<td align="char" valign="middle" char=".">0.014</td>
</tr>
<tr>
<td align="left" valign="middle">Eco3</td>
<td align="left" valign="middle"><italic>Per Capita</italic> disposable income of urban residents</td>
<td align="char" valign="middle" char=".">0.006</td>
<td align="char" valign="middle" char=".">0.023</td>
<td align="char" valign="middle" char=".">0.049</td>
</tr>
<tr>
<td align="left" valign="middle">Eco4</td>
<td align="left" valign="middle">Ratio of completed investment in industrial pollution control to GDP</td>
<td align="char" valign="middle" char=".">0.011</td>
<td align="char" valign="middle" char=".">0.01</td>
<td align="char" valign="middle" char=".">0.011</td>
</tr>
<tr>
<td align="left" valign="middle">Land1</td>
<td align="left" valign="middle">Ratio of built-up area to urban administrative area</td>
<td align="char" valign="middle" char=".">0.032</td>
<td align="char" valign="middle" char=".">0.046</td>
<td align="char" valign="middle" char=".">0.044</td>
</tr>
<tr>
<td align="left" valign="middle">Land2</td>
<td align="left" valign="middle">Green coverage rate of urban built-up areas</td>
<td align="char" valign="middle" char=".">0.051</td>
<td align="char" valign="middle" char=".">0.149</td>
<td align="char" valign="middle" char=".">0.243</td>
</tr>
<tr>
<td align="left" valign="middle">Land3</td>
<td align="left" valign="middle"><italic>Per Capita</italic> area of paved roads</td>
<td align="char" valign="middle" char=".">0.032</td>
<td align="char" valign="middle" char=".">0.034</td>
<td align="char" valign="middle" char=".">0.014</td>
</tr>
<tr>
<td align="left" valign="middle">Soc1</td>
<td align="left" valign="middle">Number of public transport vehicles in standard units per 10,000 population</td>
<td align="char" valign="middle" char=".">0.176</td>
<td align="char" valign="middle" char=".">0.036</td>
<td align="char" valign="middle" char=".">0.041</td>
</tr>
<tr>
<td align="left" valign="middle">Soc2</td>
<td align="left" valign="middle">Urban water supply coverage rate</td>
<td align="char" valign="middle" char=".">0.056</td>
<td align="char" valign="middle" char=".">0.124</td>
<td align="char" valign="middle" char=".">0.2</td>
</tr>
<tr>
<td align="left" valign="middle">Soc3</td>
<td align="left" valign="middle">Proportion of government health expenditure in total fiscal expenditure</td>
<td align="char" valign="middle" char=".">0.122</td>
<td align="char" valign="middle" char=".">0.082</td>
<td align="char" valign="middle" char=".">0.063</td>
</tr>
<tr>
<td align="left" valign="middle">Soc4</td>
<td align="left" valign="middle">Number of urban assistant practicing physicians per 10,000 population</td>
<td align="char" valign="middle" char=".">0.059</td>
<td align="char" valign="middle" char=".">0.180</td>
<td align="char" valign="middle" char=".">0.125</td>
</tr>
<tr>
<td align="left" valign="middle">Soc5</td>
<td align="left" valign="middle">Number of healthcare institution beds per 10,000 population</td>
<td align="char" valign="middle" char=".">0.044</td>
<td align="char" valign="middle" char=".">0.055</td>
<td align="char" valign="middle" char=".">0.038</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Ranking of the weights of the various elements and variables. <bold>(A)</bold> Average results. <bold>(B)</bold> GBDT (scenario 3). <bold>(C)</bold> GBDT (scenario 4). <bold>(D)</bold> XGBoost (scenario 3).</p>
</caption>
<graphic xlink:href="fpubh-13-1616841-g002.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Bar charts labeled A, B, C, and D. Chart A highlights categories Society, Land, Population, and Economy with Society having the highest value of 0.467. Charts B, C, and D display detailed bar values for demographic, economic, land, and social factors. Highest values are shown in Soc1 in chart B at 0.176, Land2 in chart C at 0.149, and Soc2 in chart D at 0.243.</alt-text>
</graphic>
</fig>
</sec>
<sec id="sec13">
<label>3.4</label>
<title>Grouped sample analysis</title>
<p>China spans a vast territory with significant geographical diversity. There are notable differences across provinces in terms of geographical location, resource endowment, industrial structure, and degree of openness. Additionally, the severe imbalance in regional development leads to regional variations in the impact of rapid urbanization on public health. Therefore, in this study, the 30 provincial-level units of Mainland China are divided into three regions, as outlined in <xref ref-type="table" rid="tab6">Table 6</xref>.</p>
<table-wrap position="float" id="tab6">
<label>Table 6</label>
<caption>
<p>Provinces included in each region.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Region</th>
<th align="left" valign="top">Provinces</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Eastern</td>
<td align="left" valign="top">Beijing, Tianjin, Hebei, Liaoning, Shandong, Shanghai, Jiangsu, Zhejiang, Fujian, Guangdong, Hainan</td>
</tr>
<tr>
<td align="left" valign="top">Central</td>
<td align="left" valign="top">Shanxi, Jilin, Heilongjiang, Anhui, Jiangxi, Henan, Hubei, Hunan</td>
</tr>
<tr>
<td align="left" valign="top">Western</td>
<td align="left" valign="top">Inner Mongolia, Guangxi, Chongqing, Sichuan, Guizhou, Yunnan, Shaanxi, Gansu, Ningxia, Qinghai, Xinjiang</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Given that the GBDT algorithm outperforms other algorithms, this section continues to use the GBDT algorithm to discuss the situation in different regions. The training set accounts for 70%, and a four-fold cross-validation method is similarly employed. The predictive performance is shown in <xref ref-type="table" rid="tab7">Table 7</xref>.</p>
<table-wrap position="float" id="tab7">
<label>Table 7</label>
<caption>
<p>Model forecasting capabilities on the three regions.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th rowspan="2">Set</th>
<th align="center" valign="top" colspan="2">Eastern</th>
<th align="center" valign="top" colspan="2">Central</th>
<th align="center" valign="top" colspan="2">Western</th>
</tr>
<tr>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
<th align="center" valign="top">RMSE</th>
<th align="center" valign="top"><italic>R</italic><sup>2</sup></th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Training Set</td>
<td align="char" valign="middle" char=".">0.003</td>
<td align="center" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.002</td>
<td align="center" valign="middle">1</td>
<td align="char" valign="middle" char=".">0.005</td>
<td align="center" valign="middle">1</td>
</tr>
<tr>
<td align="left" valign="middle">Test Set</td>
<td align="char" valign="middle" char=".">36.514</td>
<td align="center" valign="middle">0.805</td>
<td align="char" valign="middle" char=".">29.824</td>
<td align="center" valign="middle">0.628</td>
<td align="char" valign="middle" char=".">47.296</td>
<td align="center" valign="middle">0.847</td>
</tr>
</tbody>
</table>
</table-wrap>
<p><xref ref-type="table" rid="tab7">Table 7</xref> shows that there is no significant difference in the RMSE of the training sets across the three regions. In the test set, the western region has the highest RMSE, followed by the eastern region, while the central region exhibits the lowest RMSE. The <italic>R</italic><sup>2</sup> values of the training sets for all three regions are satisfactory. However, in the test set, the <italic>R</italic><sup>2</sup> for the central region is less, which may be attributed to the relatively smaller sample size in the central region.</p>
<p><xref ref-type="table" rid="tab8">Table 8</xref> presents the weights of various factors across the three major regions. In the eastern region, the most significant factor is the social factor, with a weight of 60%, followed by land factors, which account for nearly a quarter, and the lowest is the economic factor, with a weight just slightly above 5%. In the central region, economic factors dominate, with a weight exceeding 60%, followed by land factors, which account for 17.2%, while population factors have the lowest weight, under 10%. In the western region, land factors take the lead, with a weight greater than half, followed by economic factors, which account for more than one-fifth, and population factors also represent a relatively small proportion. <xref ref-type="table" rid="tab9">Table 9</xref> and <xref ref-type="fig" rid="fig3">Figure 3</xref> show the weight calculation results for the variables in the three regions, which also exhibit considerable differences.</p>
<table-wrap position="float" id="tab8">
<label>Table 8</label>
<caption>
<p>Weights of factors in the three regions.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Factor</th>
<th align="center" valign="top">Eastern</th>
<th align="center" valign="top">Central</th>
<th align="center" valign="top">Western</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Population</td>
<td align="char" valign="middle" char=".">0.103</td>
<td align="char" valign="middle" char=".">0.092</td>
<td align="char" valign="middle" char=".">0.06</td>
</tr>
<tr>
<td align="left" valign="middle">Economy</td>
<td align="char" valign="middle" char=".">0.058</td>
<td align="char" valign="middle" char=".">0.632</td>
<td align="char" valign="middle" char=".">0.214</td>
</tr>
<tr>
<td align="left" valign="middle">Land</td>
<td align="char" valign="middle" char=".">0.237</td>
<td align="char" valign="middle" char=".">0.172</td>
<td align="char" valign="middle" char=".">0.542</td>
</tr>
<tr>
<td align="left" valign="middle">Society</td>
<td align="char" valign="middle" char=".">0.602</td>
<td align="char" valign="middle" char=".">0.104</td>
<td align="char" valign="middle" char=".">0.184</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap position="float" id="tab9">
<label>Table 9</label>
<caption>
<p>The weights of the variables in the three regions.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Type</th>
<th align="left" valign="top">Variables</th>
<th align="center" valign="top">Eastern</th>
<th align="center" valign="top">Central</th>
<th align="center" valign="top">Western</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="middle">Dem1</td>
<td align="left" valign="middle">Total dependency ratio</td>
<td align="char" valign="middle" char=".">0.031</td>
<td align="char" valign="middle" char=".">0.039</td>
<td align="char" valign="middle" char=".">0.013</td>
</tr>
<tr>
<td align="left" valign="middle">Dem2</td>
<td align="left" valign="middle">Urbanization rate</td>
<td align="char" valign="middle" char=".">0.004</td>
<td align="char" valign="middle" char=".">0.012</td>
<td align="char" valign="middle" char=".">0.006</td>
</tr>
<tr>
<td align="left" valign="middle">Dem3</td>
<td align="left" valign="middle">Illiteracy rate</td>
<td align="char" valign="middle" char=".">0.003</td>
<td align="char" valign="middle" char=".">0.036</td>
<td align="char" valign="middle" char=".">0.024</td>
</tr>
<tr>
<td align="left" valign="middle">Dem4</td>
<td align="left" valign="middle">Urban population density</td>
<td align="char" valign="middle" char=".">0.065</td>
<td align="char" valign="middle" char=".">0.004</td>
<td align="char" valign="middle" char=".">0.017</td>
</tr>
<tr>
<td align="left" valign="middle">Eco1</td>
<td align="left" valign="middle">Ratio of tertiary industry value-added to secondary industry value-added</td>
<td align="char" valign="middle" char=".">0.025</td>
<td align="char" valign="middle" char=".">0.298</td>
<td align="char" valign="middle" char=".">0.147</td>
</tr>
<tr>
<td align="left" valign="middle">Eco2</td>
<td align="left" valign="middle">Foreign trade dependence</td>
<td align="char" valign="middle" char=".">0.009</td>
<td align="char" valign="middle" char=".">0.288</td>
<td align="char" valign="middle" char=".">0.053</td>
</tr>
<tr>
<td align="left" valign="middle">Eco3</td>
<td align="left" valign="middle"><italic>Per Capita</italic> disposable income of urban residents</td>
<td align="char" valign="middle" char=".">0.005</td>
<td align="char" valign="middle" char=".">0.042</td>
<td align="char" valign="middle" char=".">0.006</td>
</tr>
<tr>
<td align="left" valign="middle">Eco4</td>
<td align="left" valign="middle">Ratio of completed investment in industrial pollution control to GDP</td>
<td align="char" valign="middle" char=".">0.019</td>
<td align="char" valign="middle" char=".">0.004</td>
<td align="char" valign="middle" char=".">0.008</td>
</tr>
<tr>
<td align="left" valign="middle">Land1</td>
<td align="left" valign="middle">Ratio of built-up area to urban administrative area</td>
<td align="char" valign="middle" char=".">0.100</td>
<td align="char" valign="middle" char=".">0.062</td>
<td align="char" valign="middle" char=".">0.433</td>
</tr>
<tr>
<td align="left" valign="middle">Land2</td>
<td align="left" valign="middle">Green coverage rate of urban built-up areas</td>
<td align="char" valign="middle" char=".">0.047</td>
<td align="char" valign="middle" char=".">0.071</td>
<td align="char" valign="middle" char=".">0.095</td>
</tr>
<tr>
<td align="left" valign="middle">Land3</td>
<td align="left" valign="middle"><italic>Per Capita</italic> area of paved roads</td>
<td align="char" valign="middle" char=".">0.090</td>
<td align="char" valign="middle" char=".">0.039</td>
<td align="char" valign="middle" char=".">0.014</td>
</tr>
<tr>
<td align="left" valign="middle">Soc1</td>
<td align="left" valign="middle">Number of public transport vehicles in standard units per 10,000 population</td>
<td align="char" valign="middle" char=".">0.013</td>
<td align="char" valign="middle" char=".">0.012</td>
<td align="char" valign="middle" char=".">0.007</td>
</tr>
<tr>
<td align="left" valign="middle">Soc2</td>
<td align="left" valign="middle">Urban water supply coverage rate</td>
<td align="char" valign="middle" char=".">0.262</td>
<td align="char" valign="middle" char=".">0.004</td>
<td align="char" valign="middle" char=".">0.022</td>
</tr>
<tr>
<td align="left" valign="middle">Soc3</td>
<td align="left" valign="middle">Proportion of government health expenditure in total fiscal expenditure</td>
<td align="char" valign="middle" char=".">0.151</td>
<td align="char" valign="middle" char=".">0.005</td>
<td align="char" valign="middle" char=".">0.140</td>
</tr>
<tr>
<td align="left" valign="middle">Soc4</td>
<td align="left" valign="middle">Number of urban assistant practicing physicians per 10,000 population</td>
<td align="char" valign="middle" char=".">0.028</td>
<td align="char" valign="middle" char=".">0.012</td>
<td align="char" valign="middle" char=".">0.009</td>
</tr>
<tr>
<td align="left" valign="middle">Soc5</td>
<td align="left" valign="middle">Number of healthcare institution beds per 10,000 population</td>
<td align="char" valign="middle" char=".">0.146</td>
<td align="char" valign="middle" char=".">0.070</td>
<td align="char" valign="middle" char=".">0.006</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Ranking of the weights of all elements and variables in the three regions. <bold>(A)</bold> Weighting of elements in the three regions. <bold>(B)</bold> Eastern. <bold>(C)</bold> Central. <bold>(D)</bold> Western.</p>
</caption>
<graphic xlink:href="fpubh-13-1616841-g003.tif" mimetype="image" mime-subtype="tiff">
<alt-text content-type="machine-generated">Four bar graphs labeled A, B, C, and D compare data categories across regions. Graph A shows population, economy, land, and society data for Eastern, Central, and Western regions, with the economy highest in Central. Graphs B, C, and D present different data sets with varying values, including demographic, economic, land, and societal factors, with notable peaks such as Soc2 in B, Eco1 and Eco2 in C, and Land1 in D. Each graph displays specific numerical values on the bars.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec sec-type="discussion" id="sec14">
<label>4</label>
<title>Discussion</title>
<p>In a narrow sense, urbanization refers to the process of population concentration from rural to urban areas. In a broader sense, urbanization is often accompanied by profound changes in the economy, land use, and social structure. It is not only an economic development process but also a comprehensive manifestation of population migration, land use transformation, and social change. From a demographic perspective, urbanization is characterized by a rapid increase in the proportion of urban population relative to the total population, which serves as the primary statistical indicator for measuring the level of urbanization. Economically, urbanization drives the transformation of industrial structure, with labor shifting from agriculture-based primary industries to industry-dominated secondary industries and service-oriented tertiary industries. Industrialization plays a crucial role as a driving force for urbanization. In terms of land use, urbanization leads to changes in land utilization, with agricultural land being converted into urban construction land, thereby expanding urban spatial boundaries. At the social level, urbanization also alters the living environment, such as improvements in infrastructure and the widespread availability of public services. However, while urbanization improves residents&#x2019; well-being, it also gives rise to new social issues, such as high population density, ecological degradation, and deteriorating living conditions. This complexity and diversity of urbanization inevitably result in a multifaceted impact on infectious disease prevention and control. The influence of rapid urbanization on public health has evolved from a singular risk of population aggregation to a complex system of interactions involving environmental capacity, economic transformation, and resource allocation. Based on the weight ranking of different algorithms discussed in the previous section, the following main conclusions can be drawn.</p>
<sec id="sec15">
<label>4.1</label>
<title>Social factors as primary determinants</title>
<p>Social factors, with a weight of 0.467, rank first among all influencing factors, indicating their predominant influence on public health throughout urbanization. Social factors include healthcare and sanitation expenditures, public health facilities, and the accessibility of urban public transportation, reflecting the crucial role of infrastructure in the rapid urbanization process. Fitch (<xref ref-type="bibr" rid="ref33">33</xref>) emphasized that well-developed infrastructure is essential for preventing and controlling epidemics, as it directly affects residents&#x2019; access to healthcare resources and their capacity to cope with diseases. Indicators such as the proportion of healthcare spending in general fiscal expenditure, the number of urban practicing assistant physicians per 10,000 people, and the number of hospital beds per 10,000 people reflect the distribution of public health services during urbanization. If public health service provision can keep pace with urbanization, it can effectively alleviate the supply&#x2013;demand imbalance in healthcare and control the scope of infectious diseases. The medical resource shortages during the COVID-19 pandemic highlighted that higher healthcare investment, more specialized physicians, and broader disease prevention programs are crucial for controlling the spread of epidemics. Indicators such as the number of public transportation vehicles per 10,000 people and urban water supply coverage reflect the convenience of urban living conditions. A well-developed public transportation system can reduce the time cost for residents to access healthcare services, enhance the accessibility of medical services, and decrease the use of private cars, thereby reducing traffic congestion and emissions, and improving air quality. However, public transportation vehicles are crowded public spaces and enclosed environments, which inevitably become breeding grounds for the transmission of infectious diseases. They can also facilitate the spread of viruses from one area to another, contributing to the dissemination of diseases (<xref ref-type="bibr" rid="ref34">34</xref>). Safe drinking water supply is fundamental for preventing diseases such as enteric infections (<xref ref-type="bibr" rid="ref35">35</xref>). The 1854 cholera outbreak on Broad Street in London, during the first industrial revolution, was primarily caused by contaminated drinking water. Therefore, during urbanization, the government must prioritize the construction and improvement of the social service system. By increasing healthcare investment, optimizing public transportation infrastructure, and improving water supply coverage, the urban response capacity to public health events can be enhanced, ensuring that prevention and control measures are implemented quickly and effectively, thus controlling the spread and speed of epidemics.</p>
</sec>
<sec id="sec16">
<label>4.2</label>
<title>Land factors as structural influences</title>
<p>Land factors rank second in significance, with an average weight of 0.215, underscoring their essential role in shaping infectious disease patterns. Rapid urban growth intensifies land conversion and population agglomeration, rendering prudent spatial planning indispensable. Poorly designed land-use systems&#x2014;particularly those characterized by unbalanced development or inadequate green infrastructure&#x2014;tend to give rise to overcrowded settlements and unsanitary conditions conducive to disease transmission (<xref ref-type="bibr" rid="ref36">36</xref>). The green coverage rate in built-up areas occupies the largest proportion among land-related indicators, suggesting that a higher green coverage rate not only improves air quality but also absorbs airborne pathogens, thereby reducing the likelihood of residents contracting diseases (<xref ref-type="bibr" rid="ref8">8</xref>, <xref ref-type="bibr" rid="ref23">23</xref>, <xref ref-type="bibr" rid="ref37">37</xref>). Per capita road area reflects the smoothness of traffic flow, and smooth transportation is beneficial for the rapid passage of emergency medical vehicles, significantly shortening emergency response times and improving the efficiency of medical rescue efforts. However, it can also serve as an important medium for the spread of viruses, providing a rapid channel for viral transmission. This contradiction stems from the dual nature of the transportation system in terms of spatial mobility and social connectivity. Moreover, indicators such as the proportion of built-up areas relative to total land and per capita urban road area generally remain low in direct correlations with infection incidence, likely because these parameters influence health through longer, indirect feedback loops rather than immediate effects. Consequently, sustainable urban planning should conform to ecological principles by balancing expansion with environmental preservation. Expanding green zones, regulating urban density, and designing efficient road networks collectively enhance urban livability while constraining the spatial scope of disease diffusion. For most developing countries, particular attention should be paid to the sustainable development of suburban areas, and systemic measures should be implemented to eliminate potential sources of infection.</p>
</sec>
<sec id="sec17">
<label>4.3</label>
<title>Population factors as potential risks</title>
<p>The average weight of population factors is 0.17, exert measurable yet less direct effects compared to social and land determinants. Their influence derives primarily from the contingent risks embedded in population distribution and demographic composition. The total dependency ratio, an indicator of age structure, illustrates the socioeconomic burdens related to non-working populations. Elevated dependency ratios correspond to a higher proportion of older adults and juvenile residents&#x2014;groups inherently more susceptible to infectious diseases due to weaker immune capacity. A disproportionate presence of such vulnerable demographics intensifies the likelihood of disease spread. The proportion of urban population is an important indicator of urbanization. In the short term, the large-scale concentration of population in cities increases the frequency of person-to-person contact, accelerating the spread of infectious diseases and expanding their transmission range. If health infrastructure lags behind the speed of urbanization, it also adds pressure to the operation of existing urban infrastructure, worsening living conditions. For example, densely populated areas such as urban villages, where sanitation facilities are relatively underdeveloped, can easily become sources of infection (<xref ref-type="bibr" rid="ref38">38</xref>). However, population concentration can also facilitate the centralized provision of public health services, improving the efficiency of medical resource utilization. Illiteracy rates reflect the quality of the population. Residents with higher levels of education are generally better able to access and absorb knowledge related to infectious disease prevention and control, applying it in their daily lives, and are more likely to proactively share such knowledge. Therefore, governments should pay close attention to changes in population structure during the urbanization process, strengthen the management and health education of the mobile population, widely disseminate public health knowledge, and effectively protect the rights of vulnerable groups while ensuring the implementation of basic healthcare protection.</p>
</sec>
<sec id="sec18">
<label>4.4</label>
<title>Economic factors as underlying conditions</title>
<p>The average weight of economic factors is 0.148, possess the least direct impact on infectious disease prevalence, yet they underpin the broader environment that determines public health capacity. Economic growth elevates material living conditions and provides fiscal resources for healthcare expansion; however, without deliberate prioritization of public health infrastructure, infection rates may persist despite increased wealth. From the perspective of different indicators, the ratio of the value added by the tertiary industry to that of the secondary industry is an important measure of the optimization and upgrading of a country&#x2019;s or region&#x2019;s industrial structure. An increase in the share of the tertiary industry typically indicates a shift toward a service-oriented economy, accompanied by an improved production environment and a relative reduction in occupational hazards. For example, compared to traditional manufacturing, high-tech industries demonstrate better resource utilization efficiency and environmental friendliness. Certain sectors of the tertiary industry, closely related to residents&#x2019; health&#x2014;such as pharmaceutical research and development, healthcare services, and professional technical services&#x2014;directly reinforce societal health by promoting longevity and expanding access to scientific advancements. These sectors generally have high knowledge intensity and technical added value, effectively promoting high-quality employment, facilitating the transformation and application of scientific and technological achievements, and driving the continuous optimization and upgrading of the economic structure. This, in turn, provides solid support for the realization of the &#x201C;Healthy China&#x201D; strategic goals. The ratio of total imports and exports to GDP reflects the degree of openness. In the context of globalization, urbanization is closely linked to international trade. A higher share of imports and exports indicates frequent exchanges between cities and the outside world, which could bring advanced medical technologies but also increases the risk of cross-border transmission of infectious diseases. From July to August 2020, eight cases of COVID-19 were detected on the packaging of imported cold-chain products in Liaoning, Fujian, Jiangxi, Chongqing, Yunnan, Shandong, and Anhui, highlighting that infectious viruses can be transmitted to cities through international business travel and goods transportation (<xref ref-type="bibr" rid="ref39">39</xref>). The per capita disposable income of urban residents influences their capacity for health-related consumption. As income rises, residents are more willing and able to undergo regular check-ups, vaccinations, and early treatment, enabling &#x201C;early detection, early diagnosis, and early treatment,&#x201D; and affording better medical services and medications. The proportion of investment in industrial pollution control relative to GDP reflects the extent of industrial pollution management. Insufficient investment in this area could result in industrial pollution damaging residents&#x2019; respiratory and digestive systems, thereby increasing the incidence of infectious diseases. For example, six out of the eight major pollution incidents that occurred during the golden age of capitalism between the 1950s and 1960s were linked to industrial pollution. Although the weight of economic factors is low, they still form the material foundation for the establishment of public health and medical systems. Therefore, during the urbanization process, governments need to focus on coordinating economic development and environmental protection. By increasing investments in industrial pollution control, promoting industrial structure upgrading, and raising residents&#x2019; income levels, a solid economic foundation for public health can be created. For instance, the government could encourage businesses to adopt environmentally friendly production technologies and develop green industries to reduce the impact of industrial pollution on residents&#x2019; health.</p>
</sec>
<sec id="sec19">
<label>4.5</label>
<title>Regional differentiation in health impacts</title>
<p>In the eastern region, social factors dominate, signifying that public health outcomes rely heavily on efficient allocation of public service resources. This aligns with the characteristics of the mature urbanization stage in the east, where high population concentration forces the optimization of medical resources and infrastructure. This is reflected in the proportion of health expenditure in total fiscal expenditure and the number of healthcare institution beds per 10,000 population, both around 0.15, while urban water supply coverage rate reaches 0.262. Equalization of public services can effectively break the transmission chain of infectious diseases.</p>
<p>In the central region, economic factors exert greater weight, exposing structural tensions characteristic of mid-industrialization phases. The weight of industrial structure transformation indicators (ratio of tertiary industry value-added to secondary industry value-added) and openness indicators (ratio of foreign trade dependence) approaches 30%, driving the risk of infectious diseases. This corresponds to the &#x201C;Environmental Kuznets Curve&#x201D; theory&#x2014;during industrial transfer in the central region, lagging environmental regulations increase occupational exposure risks, making it more likely for infectious disease outbreaks.</p>
<p>In contrast, the western region is defined primarily by land-based determinants. The ecological environment in the west is relatively fragile, and excessive urban expansion may damage the original vegetation and soil structure, leading to desertification, soil erosion, and more dust storms. Dust carries bacteria, viruses, and other pathogens, which can increase the incidence of respiratory infectious diseases. Some western regions are natural plague foci, where wild animals serve as hosts or vectors for infectious diseases. When their habitats are disturbed, increased contact with humans can raise the risk of zoonotic disease transmission (<xref ref-type="bibr" rid="ref38">38</xref>). The eastern region should provide high-quality medical and health resources and improve the public health security system. The central region should alleviate the contradiction between the economy and the environment during the process of industrialization and pursue green and sustainable development. The western region should attach importance to the protection of natural ecology and curb the spread of zoonotic diseases.</p>
<p>We explicitly acknowledge slight overfitting signals in the training phase and underfitting on the test sets. The gap mainly reflects (i) small-sample, high-heterogeneity provincial panels, (ii) unobserved spatial spillovers and interactions not modeled her, and (iii) structural breaks during rapid urbanization and the COVID-19 period. Our design mitigates&#x2014;but cannot eliminate&#x2014;these risks via cross-validation, repeated shuffling, and prioritizing out-of-sample metrics; thus, results should be interpreted with caution.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec20">
<label>5</label>
<title>Conclusion</title>
<p>This study employs machine learning algorithms including Random Forest, GBDT, and XGBoost to investigate the impacts of rapid urbanization on public health, which holds significant theoretical and practical significance. Theoretically, it clarifies the hierarchical order of the influences of social, land, demographic, economic and other factors on public health, reveals the action mechanisms of each factor from multiple dimensions, and enriches the theoretical system in this field. The findings reveal a hierarchical order of influential factors: social factors emerge as the most critical determinant, followed by land-related factors, demographic factors, and economic factors. Additionally, it identifies the spatial heterogeneity in the interaction between urbanization and health, providing a new perspective for understanding how regional urbanization characteristics and ecological constraints affect public health. Practically, it offers a scientific basis for governments to formulate urbanization and public health policies, facilitates the implementation of differentiated strategies based on the dominant factors in different regions, and helps prevent and control public health risks arising from population structure changes, economic activities and other aspects in advance.</p>
<p>Regarding social dimensions, public transportation accessibility, water supply coverage, healthcare investment, and medical resource allocation directly influence population health outcomes and the accessibility of public health services. In the land-use domain, rational planning of built-up areas, green space coverage, and road infrastructure demonstrate significant health-protective effects, whereas improper land utilization may induce systemic health risks. Demographic analysis indicates that population structure shifts, spatial agglomeration patterns, and educational attainment levels present dual effects, creating opportunities for centralized public health service delivery while simultaneously amplifying infectious disease transmission risks. Economic factors exhibit paradoxical impacts: while industrial pollution control, industrial upgrading, international trade, and income growth contribute positively to public health, they concurrently harbor potential threats such as environmental degradation and cross-border transmission of infectious diseases.</p>
<p>Notably, spatial heterogeneity manifests in urbanization-health interactions across regions: social factors dominate in eastern China, economic factors prevail in central China, and land-related factors exert paramount influence in western China. This geographical stratification reflects the differential developmental priorities and ecological constraints inherent to China&#x2019;s regional urbanization trajectories.</p>
<sec id="sec21">
<label>5.1</label>
<title>Limitation</title>
<p>This study has several limitations. Due to methodological constraints, we did not account for spatial spillover effects among provincial units or interactions between different factors. Future research could be further advanced in multiple dimensions. On the one hand, it is necessary to incorporate the spatial spillover effects among provincial administrative units and the interaction effects of different factors. By constructing a more comprehensive model, this study can accurately capture inter-regional mutual influences and the combined effects of factors, thereby providing precise policy recommendations for regional coordinated development. On the other hand, further investigate the nonlinear relationships between specific variables and public health outcomes, and conduct long-term follow-up studies to observe the changing trends of impacts and long-term cumulative effects. Meanwhile, expand the research scope and sample size, integrate multidisciplinary approaches, and comprehensively apply theories and technologies from different disciplines to conduct in-depth research from multiple perspectives, so as to obtain more comprehensive and in-depth findings.</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec22">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="author-contributions" id="sec23">
<title>Author contributions</title>
<p>ZW: Software, Methodology, Supervision, Formal analysis, Data curation, Writing &#x2013; original draft. BZ: Writing &#x2013; review &#x0026; editing, Visualization, Conceptualization, Resources. LW: Investigation, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We want to thank all respondents for voluntarily participating in the survey.</p>
</ack>
<sec sec-type="COI-statement" id="sec24">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="sec25">
<title>Generative AI statement</title>
<p>The authors declare that no Gen AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="sec26">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><label>1.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Halbreich</surname><given-names>U</given-names></name></person-group>. <article-title>Impact of urbanization on mental health and well being</article-title>. <source>Curr Opin Psychiatry</source>. (<year>2023</year>) <volume>36</volume>:<fpage>200</fpage>&#x2013;<lpage>5</lpage>. doi: <pub-id pub-id-type="doi">10.1097/YCO.0000000000000864</pub-id>, PMID: <pub-id pub-id-type="pmid">36939353</pub-id></mixed-citation></ref>
<ref id="ref2"><label>2.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Collins</surname><given-names>PY</given-names></name> <name><surname>Sinha</surname><given-names>M</given-names></name> <name><surname>Concepcion</surname><given-names>T</given-names></name> <name><surname>Patton</surname><given-names>G</given-names></name> <name><surname>Way</surname><given-names>T</given-names></name> <name><surname>McCay</surname><given-names>L</given-names></name> <etal/></person-group>. <article-title>Making cities mental health friendly for adolescents and young adults</article-title>. <source>Nature</source>. (<year>2024</year>) <volume>627</volume>:<fpage>137</fpage>&#x2013;<lpage>48</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41586-023-07005-4</pub-id>, PMID: <pub-id pub-id-type="pmid">38383777</pub-id></mixed-citation></ref>
<ref id="ref3"><label>3.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cheung</surname><given-names>T</given-names></name> <name><surname>Fong</surname><given-names>KH</given-names></name> <name><surname>Xiang</surname><given-names>YT</given-names></name></person-group>. <article-title>The impact of urbanization on youth mental health in Hong Kong</article-title>. <source>Curr Opin Psychiatry</source>. (<year>2024</year>) <volume>37</volume>:<fpage>172</fpage>&#x2013;<lpage>6</lpage>. doi: <pub-id pub-id-type="doi">10.1097/YCO.0000000000000930</pub-id>, PMID: <pub-id pub-id-type="pmid">38512853</pub-id></mixed-citation></ref>
<ref id="ref4"><label>4.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Akhtar</surname><given-names>MZ</given-names></name> <name><surname>Zaman</surname><given-names>K</given-names></name> <name><surname>Khan</surname><given-names>MA</given-names></name></person-group>. <article-title>The impact of governance indicators, renewable energy demand, industrialization, and travel and transportation on urbanization: a panel study of selected Asian economies</article-title>. <source>Cities</source>. (<year>2024</year>) <volume>151</volume>:<fpage>105131</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cities.2024.105131</pub-id>, PMID: <pub-id pub-id-type="pmid">37977174</pub-id></mixed-citation></ref>
<ref id="ref5"><label>5.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dias</surname><given-names>FT</given-names></name> <name><surname>Leite</surname><given-names>ME</given-names></name> <name><surname>Fernandes</surname><given-names>EN</given-names></name> <name><surname>Cembranel</surname><given-names>P</given-names></name> <name><surname>Rita</surname><given-names>RM</given-names></name> <name><surname>de AndradeGuerra</surname><given-names>JBSO</given-names></name></person-group>. <article-title>Urban sustainability as a social function of the city: strategic correlation based on Brazilian legislation with the new urban agenda and sustainable development goals</article-title>. <source>Sustain Dev</source>. (<year>2024</year>) <volume>32</volume>:<fpage>1279</fpage>&#x2013;<lpage>90</lpage>. doi: <pub-id pub-id-type="doi">10.1002/sd.2726</pub-id></mixed-citation></ref>
<ref id="ref6"><label>6.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>W</given-names></name> <name><surname>Gu</surname><given-names>T</given-names></name> <name><surname>Xiang</surname><given-names>J</given-names></name> <name><surname>Luo</surname><given-names>T</given-names></name> <name><surname>Zeng</surname><given-names>J</given-names></name> <name><surname>Yuan</surname><given-names>Y</given-names></name></person-group>. <article-title>Ecological restoration zoning of territorial space in China: an ecosystem health perspective</article-title>. <source>J Environ Manag</source>. (<year>2024</year>) <volume>364</volume>:<fpage>121371</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jenvman.2024.121371</pub-id>, PMID: <pub-id pub-id-type="pmid">38879965</pub-id></mixed-citation></ref>
<ref id="ref7"><label>7.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hassell</surname><given-names>JM</given-names></name> <name><surname>Muloi</surname><given-names>DM</given-names></name> <name><surname>VanderWaal</surname><given-names>KL</given-names></name> <name><surname>Ward</surname><given-names>MJ</given-names></name> <name><surname>Bettridge</surname><given-names>J</given-names></name> <name><surname>Gitahi</surname><given-names>N</given-names></name> <etal/></person-group>. <article-title>Epidemiological connectivity between humans and animals across an urban landscape</article-title>. <source>Proc Natl Acad Sci</source>. (<year>2023</year>) <volume>120</volume>:<fpage>e2218860120</fpage>. doi: <pub-id pub-id-type="doi">10.1073/pnas.2218860120</pub-id>, PMID: <pub-id pub-id-type="pmid">37450494</pub-id></mixed-citation></ref>
<ref id="ref8"><label>8.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maji</surname><given-names>KJ</given-names></name> <name><surname>Dikshit</surname><given-names>AK</given-names></name> <name><surname>Arora</surname><given-names>M</given-names></name> <name><surname>Deshpande</surname><given-names>A</given-names></name></person-group>. <article-title>Estimating premature mortality attributable to PM2. 5 exposure and benefit of air pollution control policies in China for 2020</article-title>. <source>Sci Total Environ</source>. (<year>2018</year>) <volume>612</volume>:<fpage>683</fpage>&#x2013;<lpage>93</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.scitotenv.2017.08.254</pub-id>, PMID: <pub-id pub-id-type="pmid">28866396</pub-id></mixed-citation></ref>
<ref id="ref9"><label>9.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhan</surname><given-names>C</given-names></name> <name><surname>Xie</surname><given-names>M</given-names></name> <name><surname>Lu</surname><given-names>H</given-names></name> <name><surname>Liu</surname><given-names>B</given-names></name> <name><surname>Wu</surname><given-names>Z</given-names></name> <name><surname>Wang</surname><given-names>T</given-names></name> <etal/></person-group>. <article-title>Impacts of urbanization on air quality and the related health risks in a city with complex terrain</article-title>. <source>Atmos Chem Phys</source>. (<year>2023</year>) <volume>23</volume>:<fpage>771</fpage>&#x2013;<lpage>88</lpage>. doi: <pub-id pub-id-type="doi">10.5194/acp-23-771-2023</pub-id></mixed-citation></ref>
<ref id="ref10"><label>10.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Y</given-names></name> <name><surname>Su</surname><given-names>J</given-names></name> <name><surname>Liao</surname><given-names>H</given-names></name> <name><surname>Breed</surname><given-names>MF</given-names></name> <name><surname>Yao</surname><given-names>H</given-names></name> <name><surname>Shangguan</surname><given-names>H</given-names></name> <etal/></person-group>. <article-title>Increasing antimicrobial resistance and potential human bacterial pathogens in an invasive land snail driven by urbanization</article-title>. <source>Environ Sci Technol</source>. (<year>2023</year>) <volume>57</volume>:<fpage>7273</fpage>&#x2013;<lpage>84</lpage>. doi: <pub-id pub-id-type="doi">10.1021/acs.est.3c01233</pub-id>, PMID: <pub-id pub-id-type="pmid">37097110</pub-id></mixed-citation></ref>
<ref id="ref11"><label>11.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jiang</surname><given-names>TB</given-names></name> <name><surname>Deng</surname><given-names>ZW</given-names></name> <name><surname>Zhi</surname><given-names>YP</given-names></name> <name><surname>Cheng</surname><given-names>H</given-names></name> <name><surname>Gao</surname><given-names>Q</given-names></name></person-group>. <article-title>The effect of urbanization on population health: evidence from China</article-title>. <source>Front Public Health</source>. (<year>2021</year>) <volume>9</volume>:<fpage>706982</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpubh.2021.706982</pub-id>, PMID: <pub-id pub-id-type="pmid">34222193</pub-id></mixed-citation></ref>
<ref id="ref12"><label>12.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kadakia</surname><given-names>KT</given-names></name> <name><surname>Galea</surname><given-names>S</given-names></name></person-group>. <article-title>Urbanization and the future of population health</article-title>. <source>Milbank Q</source>. (<year>2023</year>) <volume>101</volume>:<fpage>153</fpage>&#x2013;<lpage>75</lpage>. doi: <pub-id pub-id-type="doi">10.1111/1468-0009.12624</pub-id>, PMID: <pub-id pub-id-type="pmid">37096620</pub-id></mixed-citation></ref>
<ref id="ref13"><label>13.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tay</surname><given-names>DA</given-names></name> <name><surname>Ocansey</surname><given-names>RT</given-names></name></person-group>. <article-title>Impact of urbanization on health and well-being in Ghana. Status of research, intervention strategies and future directions: a rapid review</article-title>. <source>Front Public Health</source>. (<year>2022</year>) <volume>10</volume>:<fpage>877920</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpubh.2022.877920</pub-id>, PMID: <pub-id pub-id-type="pmid">35836994</pub-id></mixed-citation></ref>
<ref id="ref14"><label>14.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>X</given-names></name> <name><surname>Yao</surname><given-names>W</given-names></name> <name><surname>Luo</surname><given-names>Q</given-names></name> <name><surname>Yun</surname><given-names>J</given-names></name></person-group>. <article-title>Spatial relationship between ecosystem health and urbanization in coastal mountain city, Qingdao, China</article-title>. <source>Ecol Inform</source>. (<year>2024</year>) <volume>79</volume>:<fpage>102458</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ecoinf.2023.102458</pub-id></mixed-citation></ref>
<ref id="ref15"><label>15.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname><given-names>D</given-names></name> <name><surname>Li</surname><given-names>X</given-names></name> <name><surname>Yu</surname><given-names>J</given-names></name> <name><surname>Shi</surname><given-names>X</given-names></name> <name><surname>Liu</surname><given-names>P</given-names></name> <name><surname>Tian</surname><given-names>P</given-names></name></person-group>. <article-title>Whether urbanization has intensified the spread of infectious diseases&#x2014;renewed question by the COVID-19 pandemic</article-title>. <source>Front Public Health</source>. (<year>2021</year>) <volume>9</volume>:<fpage>699710</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpubh.2021.699710</pub-id></mixed-citation></ref>
<ref id="ref16"><label>16.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>XP</given-names></name> <name><surname>Yu</surname><given-names>DS</given-names></name></person-group>. <article-title>Does urbanization aggravate the interregional transmission of infectious diseases? Analysis based on spatial spillover</article-title>. <source>Econ Sci</source>. (<year>2022</year>) <volume>251</volume>:<fpage>107</fpage>&#x2013;<lpage>19</lpage>.</mixed-citation></ref>
<ref id="ref17"><label>17.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname><given-names>S</given-names></name> <name><surname>Liu</surname><given-names>Y</given-names></name> <name><surname>Fang</surname><given-names>Y</given-names></name></person-group>. <article-title>Measuring the differences of public health service facilities and their influencing factors</article-title>. <source>Land</source>. (<year>2021</year>) <volume>10</volume>:<fpage>1225</fpage>. doi: <pub-id pub-id-type="doi">10.3390/land10111225</pub-id></mixed-citation></ref>
<ref id="ref18"><label>18.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qiu</surname><given-names>YZ</given-names></name> <name><surname>Chen</surname><given-names>HS</given-names></name> <name><surname>Li</surname><given-names>ZG</given-names></name> <name><surname>Wang</surname><given-names>RY</given-names></name> <name><surname>Liu</surname><given-names>Y</given-names></name> <name><surname>Qin</surname><given-names>XF</given-names></name></person-group>. <article-title>Exploring neighborhood environmental effects on mental health: a case study in Guangzhou, China</article-title>. <source>Prog Geogr</source>. (<year>2019</year>) <volume>38</volume>:<fpage>283</fpage>&#x2013;<lpage>95</lpage>. doi: <pub-id pub-id-type="doi">10.18306/dlkxjz.2019.02.011</pub-id></mixed-citation></ref>
<ref id="ref19"><label>19.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cengiz</surname><given-names>D</given-names></name> <name><surname>Dube</surname><given-names>A</given-names></name> <name><surname>Lindner</surname><given-names>A</given-names></name> <name><surname>Zentler-Munro</surname><given-names>D</given-names></name></person-group>. <article-title>Seeing beyond the trees: using machine learning to estimate the impact of minimum wages on labor market outcomes</article-title>. <source>J Labor Econ</source>. (<year>2022</year>) <volume>40</volume>:<fpage>203</fpage>&#x2013;<lpage>47</lpage>. doi: <pub-id pub-id-type="doi">10.1086/718497</pub-id></mixed-citation></ref>
<ref id="ref20"><label>20.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname><given-names>QQ</given-names></name> <name><surname>Zhong</surname><given-names>WZ</given-names></name></person-group>. <article-title>Does urbanization promote public health?</article-title> <source>Econ Surv</source>. (<year>2018</year>) <volume>35</volume>:<fpage>127</fpage>&#x2013;<lpage>34</lpage>. doi: <pub-id pub-id-type="doi">10.15931/j.cnki.1006-1096.20180925.005</pub-id></mixed-citation></ref>
<ref id="ref21"><label>21.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>YB</given-names></name> <name><surname>Zhang</surname><given-names>Y</given-names></name> <name><surname>Li</surname><given-names>JG</given-names></name></person-group>. <article-title>Digital finance and carbon emissions: an empirical test based on micro data and machine learning model</article-title>. <source>Chin Popul Resour Environ</source>. (<year>2022</year>) <volume>32</volume>:<fpage>1</fpage>&#x2013;<lpage>11</lpage>.</mixed-citation></ref>
<ref id="ref22"><label>22.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname><given-names>ZY</given-names></name> <name><surname>Chen</surname><given-names>K</given-names></name> <name><surname>Chen</surname><given-names>XL</given-names></name> <name><surname>Chen</surname><given-names>YB</given-names></name></person-group>. <article-title>Identifying the influencing factors of inflation: reexamination based on machine learning methods</article-title>. <source>Statistical Research</source>. (<year>2022</year>) <volume>39</volume>:<fpage>132</fpage>&#x2013;<lpage>47</lpage>. doi: <pub-id pub-id-type="doi">10.19343/j.cnki.11-1302/c.2022.06.009</pub-id></mixed-citation></ref>
<ref id="ref23"><label>23.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>X</given-names></name> <name><surname>Han</surname><given-names>L</given-names></name> <name><surname>Wei</surname><given-names>H</given-names></name> <name><surname>Tan</surname><given-names>X</given-names></name> <name><surname>Zhou</surname><given-names>W</given-names></name> <name><surname>Li</surname><given-names>W</given-names></name> <etal/></person-group>. <article-title>Linking urbanization and air quality together: a review and a perspective on the future sustainable urban development</article-title>. <source>J Clean Prod</source>. (<year>2022</year>) <volume>346</volume>:<fpage>130988</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jclepro.2022.130988</pub-id></mixed-citation></ref>
<ref id="ref24"><label>24.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Giglio</surname><given-names>S</given-names></name> <name><surname>Kelly</surname><given-names>B</given-names></name> <name><surname>Xiu</surname><given-names>D</given-names></name></person-group>. <article-title>Factor models, machine learning, and asset pricing</article-title>. <source>Annu Rev Financ Econ</source>. (<year>2022</year>) <volume>14</volume>:<fpage>337</fpage>&#x2013;<lpage>68</lpage>. doi: <pub-id pub-id-type="doi">10.1146/annurev-financial-101521-104735</pub-id></mixed-citation></ref>
<ref id="ref25"><label>25.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chinco</surname><given-names>A</given-names></name> <name><surname>Clark-Joseph</surname><given-names>AD</given-names></name> <name><surname>Ye</surname><given-names>M</given-names></name></person-group>. <article-title>Sparse signals in the cross-section of returns</article-title>. <source>J Finance</source>. (<year>2019</year>) <volume>74</volume>:<fpage>449</fpage>&#x2013;<lpage>92</lpage>. doi: <pub-id pub-id-type="doi">10.1111/jofi.12733</pub-id></mixed-citation></ref>
<ref id="ref26"><label>26.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gu</surname><given-names>S</given-names></name> <name><surname>Kelly</surname><given-names>B</given-names></name> <name><surname>Xiu</surname><given-names>D</given-names></name></person-group>. <article-title>Empirical asset pricing via machine learning</article-title>. <source>Rev Financ Stud</source>. (<year>2020</year>) <volume>33</volume>:<fpage>2223</fpage>&#x2013;<lpage>73</lpage>. doi: <pub-id pub-id-type="doi">10.1093/rfs/hhaa009</pub-id></mixed-citation></ref>
<ref id="ref27"><label>27.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Athey</surname><given-names>S</given-names></name> <name><surname>Imbens</surname><given-names>GW</given-names></name></person-group>. <article-title>Machine learning methods that economists should know about</article-title>. <source>Annu Rev Econ</source>. (<year>2019</year>) <volume>11</volume>:<fpage>685</fpage>&#x2013;<lpage>725</lpage>. doi: <pub-id pub-id-type="doi">10.1146/annurev-economics-080217-053433</pub-id></mixed-citation></ref>
<ref id="ref28"><label>28.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Freund</surname><given-names>Y</given-names></name> <name><surname>Schapire</surname><given-names>RE</given-names></name></person-group>. <article-title>A decision-theoretic generalization of on-line learning and an application to boosting</article-title>. <source>J Comput Syst Sci</source>. (<year>1997</year>) <volume>55</volume>:<fpage>119</fpage>&#x2013;<lpage>39</lpage>. doi: <pub-id pub-id-type="doi">10.1006/jcss.1997.1504</pub-id></mixed-citation></ref>
<ref id="ref29"><label>29.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krauss</surname><given-names>C</given-names></name> <name><surname>Do</surname><given-names>XA</given-names></name> <name><surname>Huck</surname><given-names>N</given-names></name></person-group>. <article-title>Deep neural networks, gradient-boosted trees, random forests: statistical arbitrage on the S&#x0026;P 500</article-title>. <source>Eur J Oper Res</source>. (<year>2017</year>) <volume>259</volume>:<fpage>689</fpage>&#x2013;<lpage>702</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.ejor.2016.10.031</pub-id></mixed-citation></ref>
<ref id="ref30"><label>30.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname><given-names>L.</given-names></name></person-group> <article-title>Statistical Modeling: The Two Cultures (with comments and a rejoinder by the author)</article-title>. <source>Statist. Sci.</source> (<year>2001</year>)<volume>16</volume>:<fpage>199</fpage>&#x2013;<lpage>231</lpage>. doi: <pub-id pub-id-type="doi">10.1214/ss/1009213726</pub-id></mixed-citation></ref>
<ref id="ref31"><label>31.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rigatti</surname><given-names>SJ</given-names></name></person-group>. <article-title>Random forest</article-title>. <source>J Insur Med</source>. (<year>2017</year>) <volume>47</volume>:<fpage>31</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.17849/insm-47-01-31-39.1</pub-id>, PMID: <pub-id pub-id-type="pmid">28836909</pub-id></mixed-citation></ref>
<ref id="ref32"><label>32.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>T</given-names></name> <name><surname>Guestrin</surname><given-names>C</given-names></name></person-group>. <article-title>Xgboost: a scalable tree boosting system</article-title>. <conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>. (<year>2016</year>): <fpage>785</fpage>&#x2013;<lpage>794</lpage>.</mixed-citation></ref>
<ref id="ref33"><label>33.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fitch</surname><given-names>JP</given-names></name></person-group>. <article-title>Engineering a global response to infectious diseases</article-title>. <source>Proc IEEE</source>. (<year>2015</year>) <volume>103</volume>:<fpage>263</fpage>&#x2013;<lpage>72</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JPROC.2015.2389146</pub-id>, PMID: <pub-id pub-id-type="pmid">34191866</pub-id></mixed-citation></ref>
<ref id="ref34"><label>34.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname><given-names>Y</given-names></name> <name><surname>Li</surname><given-names>C</given-names></name> <name><surname>Dong</surname><given-names>H</given-names></name> <name><surname>Wang</surname><given-names>Z</given-names></name> <name><surname>Martinez</surname><given-names>L</given-names></name> <name><surname>Sun</surname><given-names>Z</given-names></name> <etal/></person-group>. <article-title>Community outbreak investigation of SARS-CoV-2 transmission among bus riders in eastern China</article-title>. <source>JAMA Intern Med</source>. (<year>2020</year>) <volume>180</volume>:<fpage>1665</fpage>&#x2013;<lpage>71</lpage>. doi: <pub-id pub-id-type="doi">10.1001/jamainternmed.2020.522</pub-id>, PMID: <pub-id pub-id-type="pmid">32870239</pub-id></mixed-citation></ref>
<ref id="ref35"><label>35.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Leandro</surname><given-names>J</given-names></name> <name><surname>Hotta</surname><given-names>CI</given-names></name> <name><surname>Pinto</surname><given-names>TA</given-names></name> <name><surname>Ahadzie</surname><given-names>DK</given-names></name></person-group>. <article-title>Expected annual probability of infection: a flood-risk approach to waterborne infectious diseases</article-title>. <source>Water Res</source>. (<year>2022</year>) <volume>219</volume>:<fpage>118561</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.watres.2022.118561</pub-id>, PMID: <pub-id pub-id-type="pmid">35576764</pub-id></mixed-citation></ref>
<ref id="ref36"><label>36.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Matthew</surname><given-names>R</given-names></name> <name><surname>Chiotha</surname><given-names>S</given-names></name> <name><surname>Orbinski</surname><given-names>J</given-names></name> <name><surname>Talukder</surname><given-names>B</given-names></name></person-group>. <article-title>Research note: climate change, peri-urban space and emerging infectious disease</article-title>. <source>Landsc Urban Plan</source>. (<year>2022</year>) <volume>218</volume>:<fpage>104298</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.landurbplan.2021.104298</pub-id></mixed-citation></ref>
<ref id="ref37"><label>37.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>P</given-names></name> <name><surname>Hanlon</surname><given-names>B</given-names></name></person-group>. <article-title>Urban green space, respiratory health and rising temperatures: an examination of the complex relationship between green space and adult asthma across racialized neighborhoods in Los Angeles County</article-title>. <source>Landsc Urban Plann</source>. (<year>2025</year>) <volume>258</volume>:<fpage>105320</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.landurbplan.2025.105320</pub-id></mixed-citation></ref>
<ref id="ref38"><label>38.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hassell</surname><given-names>JM</given-names></name> <name><surname>Begon</surname><given-names>M</given-names></name> <name><surname>Ward</surname><given-names>MJ</given-names></name> <name><surname>F&#x00E8;vre</surname><given-names>EM</given-names></name></person-group>. <article-title>Urbanization and disease emergence: dynamics at the wildlife&#x2013;livestock&#x2013;human interface</article-title>. <source>Trends Ecol Evol</source>. (<year>2017</year>) <volume>32</volume>:<fpage>55</fpage>&#x2013;<lpage>67</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tree.2016.09.012</pub-id>, PMID: <pub-id pub-id-type="pmid">28029378</pub-id></mixed-citation></ref>
<ref id="ref39"><label>39.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Montiel</surname><given-names>I</given-names></name> <name><surname>Park</surname><given-names>J</given-names></name> <name><surname>Husted</surname><given-names>BW</given-names></name> <name><surname>Velez-Calle</surname><given-names>A</given-names></name></person-group>. <article-title>Tracing the connections between international business and communicable diseases</article-title>. <source>J Int Bus Stud</source>. (<year>2022</year>) <volume>53</volume>:<fpage>1785</fpage>. doi: <pub-id pub-id-type="doi">10.1057/s41267-022-00512-y</pub-id>, PMID: <pub-id pub-id-type="pmid">35345569</pub-id></mixed-citation></ref>
<ref id="ref40"><label>40.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Z</given-names></name> <name><surname>Zhao</surname><given-names>M</given-names></name> <name><surname>Zhang</surname><given-names>Y</given-names></name> <name><surname>Feng</surname><given-names>Y</given-names></name></person-group>. <article-title>How does urbanization affect public health? New evidence from 175 countries worldwide</article-title>. <source>Front Public Health</source>. (<year>2022</year>) <volume>10</volume>:<fpage>1096964</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpubh.2022.1096964</pub-id>, PMID: <pub-id pub-id-type="pmid">36684862</pub-id></mixed-citation></ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2738149/overview">Md Galal Uddin</ext-link>, University of Galway, Ireland</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1957718/overview">Augustin F. C. Holl</ext-link>, Xiamen University, China</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2931252/overview">Ran Tong</ext-link>, The University of Texas at Dallas, United States</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3152401/overview">Ruobing Bai</ext-link>, AstraZeneca, United States</p>
</fn>
</fn-group>
</back>
</article>