<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Future Transp.</journal-id>
<journal-title>Frontiers in Future Transportation</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Future Transp.</abbrev-journal-title>
<issn pub-type="epub">2673-5210</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1603726</article-id>
<article-id pub-id-type="doi">10.3389/ffutr.2025.1603726</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Future Transportation</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Adaptive vehicle routing for humanitarian aid in conflict-affected regions: a practitioner-informed deep reinforcement learning approach</article-title>
<alt-title alt-title-type="left-running-head">Mili and Argoubi</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/ffutr.2025.1603726">10.3389/ffutr.2025.1603726</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Mili</surname>
<given-names>Khaled</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2964786/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Argoubi</surname>
<given-names>Majdi</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3012211/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Quantitative Methods, College of Business, King Faisal University</institution>, <addr-line>Al-Ahsa</addr-line>, <country>Saudi Arabia</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Quantitative Methods, University of Sousse</institution>, <addr-line>Sousse</addr-line>, <country>Tunisia</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2420313/overview">Mohsen Momenitabar</ext-link>, University of South Florida, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1428196/overview">Teddy Lazebnik</ext-link>, University of Haifa, Israel</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2341198/overview">Adis Alihod&#x17e;i&#x107;</ext-link>, University of Sarajevo, Bosnia and Herzegovina</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Khaled Mili, <email>kmili@kfu.edu.sa</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>24</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>6</volume>
<elocation-id>1603726</elocation-id>
<history>
<date date-type="received">
<day>31</day>
<month>03</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>09</day>
<month>09</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Mili and Argoubi.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Mili and Argoubi</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Humanitarian aid delivery in conflict-affected regions faces significant challenges due to dynamic security risks, uncertain demand, and complex operational constraints. Traditional optimization methods struggle with computational intractability and lack adaptability for real-time decision-making in volatile environments. To address these limitations, we propose a novel hybrid framework that integrates Deep Reinforcement Learning (DRL) with Graph Neural Networks (GNNs) and deterministic constraint validation, informed by practitioner insights to ensure real-world applicability. Our approach employs Proximal Policy Optimization (PPO) enhanced by GNN-based spatial representations to learn adaptive, efficient vehicle routing policies under uncertainty. A post-decision validation mechanism enforces feasibility by penalizing constraint violations based on a deterministic equivalent model. We evaluate our method on realistic, georeferenced datasets reflecting Afghan road networks and conflict data, comparing it against classical PPO and heuristic baselines. Results demonstrate that PPO-GNN significantly reduces operational costs (by 7.9%), security risk exposure (by 15.2%), and unmet demand, while improving reliability and adherence to constraints. The approach scales effectively across network sizes and maintains robustness under stochastic variations in demand and security conditions. Our framework balances computational efficiency with practical relevance, aligning with humanitarian priorities and offering a promising decision-support tool for aid logistics in conflict zones.</p>
</abstract>
<kwd-group>
<kwd>humanitarian aid delivery</kwd>
<kwd>conflict zone logistics</kwd>
<kwd>deep reinforcement learning</kwd>
<kwd>graph neural networks</kwd>
<kwd>proximal policy optimization</kwd>
<kwd>adaptive vehicle routing</kwd>
<kwd>practitioner-informed modeling</kwd>
</kwd-group>
<contract-sponsor id="cn001">Deanship of Scientific Research, King Faisal University<named-content content-type="fundref-id">10.13039/501100004686</named-content>
</contract-sponsor>
<counts>
<page-count count="17"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Freight Transport and Logistics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Optimizing humanitarian aid operations in conflict-affected regions involves managing uncertainty in security conditions, route accessibility, and resource demands, making it a complex, high-dimensional problem. Traditional optimization methods provide structured formulations but often fail to handle real-world stochastic variations and rapidly changing conditions, leading to infeasible or suboptimal solutions. Moreover, the high computational complexity of exact solvers makes them impractical for real-time decision-making in crisis situations.</p>
<p>Recent studies have shown that significant challenges for planning and executing humanitarian operations globally remain, with limited evidence of successful implementation of optimization models in the field. This implementation gap is directly linked to practitioners&#x2019; trust in the solution models, which is undermined using unrealistic assumptions, oversimplification of operational complexities, and time-consuming solution methods that are impractical for field operations. Critically, studies have revealed that only approximately 10% of humanitarian logistics models incorporate practitioner input in their design process, creating a substantial misalignment between academic objectives and field priorities.</p>
<p>To systematically address both the stochastic nature and computational challenges of humanitarian aid delivery in conflict zones, we first develop a stochastic mathematical model that accurately captures real-world uncertainties. In this formulation, each aid delivery location <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> corresponds to a distribution center <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> with a demand modeled as a random variable with mean <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and standard deviation <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. While this stochastic formulation effectively represents aid requirement variability, solving it directly is computationally intractable due to probabilistic constraints and the need for extensive scenario evaluation.</p>
<p>To enable practical computation, we derive a deterministic equivalent formulation, approximating the stochastic problem using expected values and chance constraints. This transformation makes the problem more tractable and provides a structured benchmark for evaluating solution feasibility. However, despite this reformulation, the deterministic model remains computationally prohibitive for large-scale networks due to its combinatorial nature and the necessity for exhaustive enumeration. Moreover, in conflict zones, conditions change rapidly, requiring solutions that can adapt in near real-time.</p>
<p>To overcome these limitations, we propose a hybrid optimization framework that integrates Deep Reinforcement Learning (DRL) enhanced with Graph Neural Networks (GNNs), while incorporating the deterministic model for solution validation and refinement. Our approach builds upon Proximal Policy Optimization (PPO), a state-of-the-art DRL algorithm that learns optimized vehicle assignment and routing decisions through interaction with a simulated humanitarian aid delivery environment. Additionally, we employ GNNs to extract spatial dependencies from the delivery network, enriching the DRL agent&#x2019;s state representation and improving decision-making under uncertainty. Finally, we introduce a Constraint Validation Mechanism that leverages the deterministic model as a feasibility check to refine learned policies, ensuring compliance with operational constraints such as vehicle capacity limits, security thresholds, and delivery time windows.</p>
<p>Unlike traditional approaches, our hybrid methodology does not entirely discard deterministic optimization. Instead, it integrates deterministic validation as a post-decision refinement step, ensuring that the DRL-generated solutions remain feasible. By combining DRL&#x2019;s adaptability, GNN&#x2019;s representation power, and deterministic feasibility checks, our method achieves scalable, high-quality solutions for real-world humanitarian aid operations in volatile environments. Furthermore, through structured engagement with field practitioners, we ensure our model aligns with operational priorities identified in humanitarian contexts&#x2014;prioritizing service reliability, quality, and operational security rather than purely focusing on cost minimization as commonly found in academic literature.</p>
<p>The key contributions of this paper are as follows:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> We develop a rigorous stochastic mathematical model for humanitarian aid delivery optimization in conflict zones, along with its deterministic equivalent, enabling structured feasibility validation under uncertainty.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> We propose a hybrid methodology that integrates PPO-based Deep Reinforcement Learning, Graph Neural Networks, and deterministic constraint validation, facilitating scalable, adaptive, and robust decision-making in volatile and dynamic environments.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> We create a realistic simulation environment based on georeferenced road networks and conflict data from affected regions, accurately modeling humanitarian aid distribution networks to enable robust policy evaluation.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> We incorporate practitioner perspectives from published literature to align our model objectives, reward functions, and constraints with real operational priorities, enhancing practical relevance and trustworthiness (<xref ref-type="bibr" rid="B29">Rodr&#xed;guez-Esp&#xed;ndola et al., 2023</xref>; <xref ref-type="bibr" rid="B11">Holgu&#xed;n-Veras et al., 2013</xref>).</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> We conduct extensive experimental benchmarking against deterministic solvers, classical DRL methods, and heuristic-based approaches, demonstrating superior performance in cost efficiency, demand fulfillment, operational feasibility, and compliance with practitioner-informed constraints (<xref ref-type="bibr" rid="B7">Clarke and Wright, 1964</xref>; <xref ref-type="bibr" rid="B30">Schulman et al., 2017</xref>).</p>
</list-item>
</list>
</p>
<p>The remainder of this paper is structured as follows. <xref ref-type="sec" rid="s2">Section 2</xref> reviews the literature on humanitarian logistics, stochastic routing, and hybrid AI-based approaches, highlighting existing methods and their limitations. <xref ref-type="sec" rid="s3">Section 3</xref> presents the problem definition and mathematical models, first introducing the stochastic formulation of the aid delivery problem, followed by its deterministic equivalent for feasibility validation. <xref ref-type="sec" rid="s4">Section 4</xref> details our proposed hybrid methodology, which integrates PPO-based DRL, GNN-enhanced state representation, and deterministic constraint validation to generate scalable and operationally feasible solutions. <xref ref-type="sec" rid="s5">Section 5</xref> evaluates PPO-GNN&#x2019;s performance against classical PPO without GNN and the Clarke-Wright Savings Algorithm, focusing on solution quality, computational efficiency, and robustness to stochastic variations and security disruptions. Finally, <xref ref-type="sec" rid="s6">Section 6</xref> concludes the paper by summarizing key findings and outlining future research directions for improving computational efficiency and real-time adaptability.</p>
</sec>
<sec id="s2">
<title>2 Literature review</title>
<p>This section reviews the existing literature on humanitarian logistics, approaches for handling uncertainty in conflict zones, and hybrid AI methods that integrate optimization with deep reinforcement learning (DRL) and graph neural networks (GNNs). We subsequently identify the gaps that our work aims to address, providing a foundation for our contributions.</p>
<p>The vehicle routing problem (VRP) has been extensively studied in operations research, with the classic Capacitated Vehicle Routing Problem (CVRP) serving as a foundational model for numerous real-world applications (<xref ref-type="bibr" rid="B19">Laporte, 1992</xref>). Within the context of humanitarian aid delivery in conflict zones, the problem becomes significantly more complex due to factors such as dynamic security conditions, heterogeneous fleets, high-priority demands, and operational constraints including security thresholds and checkpoint-based route structures (<xref ref-type="bibr" rid="B25">&#xd6;zdamar and Ertem, 2015</xref>; <xref ref-type="bibr" rid="B1">Altay and Green, 2006</xref>). Various VRP variants, such as the split delivery VRP (SDVRP), have been proposed to address scenarios where demand at a single node can be fulfilled by multiple vehicles (<xref ref-type="bibr" rid="B8">Dror et al., 1989</xref>). These models have been subsequently extended to incorporate humanitarian-specific constraints, including priority-based scheduling, security risk minimization, and checkpoint traversal requirements (<xref ref-type="bibr" rid="B14">Huang et al., 2012</xref>).</p>
<p>For instance, <xref ref-type="bibr" rid="B2">Balcik et al. (2008)</xref> introduced comprehensive models for humanitarian logistics incorporating time windows and heterogeneous fleets, while <xref ref-type="bibr" rid="B25">&#xd6;zdamar and Ertem (2015)</xref> proposed hybrid heuristics for solving large-scale aid delivery problems with multiple depots. However, most existing studies on humanitarian logistics operate under deterministic parameter assumptions, such as fixed demand and security conditions, which significantly limits their applicability in real-world conflict scenarios where uncertainty in demand, route accessibility, and security risks is prevalent (<xref ref-type="bibr" rid="B23">Najafi et al., 2013</xref>; <xref ref-type="bibr" rid="B28">Rodr&#xed;guez-Esp&#xed;ndola et al., 2018</xref>).</p>
<p>Recent assessments of disaster management have revealed that several challenges for planning and executing operations globally persist, suggesting a limited level of implementation of advances in humanitarian logistics (<xref ref-type="bibr" rid="B24">Negi, 2022</xref>). Studies have identified that implementation success is directly linked to practitioners&#x2019; trust in research findings, which is often undermined by unrealistic assumptions in optimization models (<xref ref-type="bibr" rid="B10">Galindo and Batta, 2013</xref>). A critical analysis of humanitarian logistics optimization models by <xref ref-type="bibr" rid="B29">Rodr&#xed;guez-Esp&#xed;ndola et al. (2023)</xref> revealed that only approximately 10% incorporate input from practitioners in their modeling decisions. This disconnect helps explain why most academic models prioritize cost minimization, while multi-criteria decision analysis with practitioners revealed that reliability, quality of service, and prioritization of most affected areas rank significantly higher than cost in real operations.</p>
<p>Uncertainty in routing problems has been systematically addressed through stochastic programming and robust optimization techniques. Chance-constrained methods ensure that constraints such as demand satisfaction are met with a specified probability (<xref ref-type="bibr" rid="B23">Najafi et al., 2013</xref>). Recourse strategies, conversely, introduce penalty costs for deviations from planned routes, allowing for adaptive decision-making in response to realized uncertainties (<xref ref-type="bibr" rid="B27">P&#xe9;rez-Rodr&#xed;guez and Holgu&#xed;n-Veras, 2016</xref>). For example, chance-constrained formulations have been applied to VRPs with stochastic demand, where the objective is to minimize costs while ensuring that the probability of route failure, such as exceeding vehicle capacity, remains below a predetermined threshold (<xref ref-type="bibr" rid="B6">Bozorgi-Amiri et al., 2013</xref>). Recourse models, including the two-stage stochastic VRP, have been utilized to optimize initial routing decisions while accounting for the cost of adjusting routes based on observed demand (<xref ref-type="bibr" rid="B12">Hu et al., 2019</xref>). Despite these significant advancements, applying these methods to large-scale humanitarian aid delivery problems in conflict zones remains challenging due to their computational complexity and the need for scalable solutions that can adapt to rapidly changing conditions (<xref ref-type="bibr" rid="B34">Wex et al., 2014</xref>).</p>
<p>Given the computational demands and limited real-time adaptability of stochastic optimization techniques, alternative approaches are necessary. This is where artificial intelligence (AI) techniques, particularly DRL and GNNs, become crucial for addressing these limitations.</p>
<p>Deep reinforcement learning has emerged as a powerful tool for sequential decision-making, particularly in dynamic and uncertain environments. Unlike earlier policy gradient methods such as REINFORCE and Advantage Actor-Critic (A2C), Proximal Policy Optimization (PPO) introduces a clipped objective function that prevents drastic policy updates, resulting in more stable training (<xref ref-type="bibr" rid="B30">Schulman et al., 2017</xref>). This characteristic makes PPO particularly well-suited for large-scale routing problems in conflict zones, where decision spaces are high-dimensional and exploration is critical. PPO has gained widespread popularity due to its sample efficiency, stability, and ability to handle high-dimensional state and action spaces, making it an ideal choice for problems like humanitarian aid delivery optimization, where agents must learn policies that balance exploration and exploitation while adhering to operational constraints.</p>
<p>PPO has been successfully applied to various logistics and routing problems, demonstrating its ability to handle combinatorial decision-making efficiently. <xref ref-type="bibr" rid="B3">Bello et al. (2016)</xref> integrated PPO with a pointer network for vehicle routing, achieving significant improvements in computational efficiency compared to traditional solvers. Similarly, <xref ref-type="bibr" rid="B15">Kool et al. (2019)</xref> explored PPO for dynamic routing, showing that it adapts effectively to changes in demand and network conditions. Recent work by <xref ref-type="bibr" rid="B5">Bogyrbayeva et al. (2022)</xref> has also applied PPO to routing problems, demonstrating its effectiveness in learning adaptive policies for complex decision-making tasks. However, these applications often lack mechanisms for ensuring feasibility and scalability in large-scale networks, which motivates the integration of PPO with complementary techniques, such as GNNs and constraint validation.</p>
<p>Graph neural networks have gained significant traction in routing problems due to their ability to model complex spatial and relational structures (<xref ref-type="bibr" rid="B35">Wu et al., 2020</xref>). GNNs can effectively capture the underlying topology of routing networks, enabling more sophisticated feature extraction and decision-making capabilities. Recent research has explored the integration of GNNs with DRL for solving VRPs, combining the strengths of both approaches to achieve state-of-the-art performance. For instance, <xref ref-type="bibr" rid="B5">Bogyrbayeva et al. (2022)</xref> developed a GNN-based encoder for the VRP, which was combined with a DRL decoder to generate high-quality solutions. Similarly, <xref ref-type="bibr" rid="B21">Li et al. (2020)</xref> proposed a hybrid DRL-GNN framework for the dynamic VRP, demonstrating its ability to adapt to changing environments in real-time. Despite these promising developments, the application of hybrid DRL-GNN methods to humanitarian aid delivery problems in conflict zones remains significantly underexplored. Existing studies often focus on deterministic settings or fail to account for the unique challenges of humanitarian logistics, such as the critical need for real-time adaptability and the handling of high-dimensional state spaces with security constraints.</p>
<p>Recent advances in deep reinforcement learning have demonstrated increasingly promising results for routing and navigation problems, particularly when incorporating domain-specific knowledge and multi-agent coordination strategies. Adaptive search methods for time-dependent vehicle routing problems exemplify this trend by emphasizing the integration of environmental and temporal dynamics, which directly aligns with the challenges inherent in dynamic and uncertain humanitarian logistics contexts (<xref ref-type="bibr" rid="B36">Yue et al., 2024</xref>). Similarly, cooperative multi-agent RL approaches have been successfully developed for complex dynamic assignments, illustrating the potential for extending hybrid frameworks to multi-vehicle or multi-stakeholder operational settings (<xref ref-type="bibr" rid="B22">Merkulov et al., 2025</xref>).</p>
<p>Models addressing collective motion and collision avoidance through reinforcement learning have highlighted effective strategies for decentralized navigation and safety constraint enforcement, which parallels the risk-avoidance mechanisms and operational limitations addressed in our deterministic validation module (<xref ref-type="bibr" rid="B17">Krongauz and Lazebnik, 2023</xref>). Furthermore, physics-informed deep RL has been successfully applied to conflict resolution in safety-critical environments, demonstrating the substantial benefits of embedding domain constraints within learning architectures to improve both robustness and compliance with hard operational constraints (<xref ref-type="bibr" rid="B37">Zhao and Liu, 2021</xref>).</p>
<p>In the domain of resource allocation problems under uncertainty, agent-based simulations combined with deep RL have proven particularly effective in dynamic and complex operational contexts that closely resemble humanitarian aid distribution scenarios (<xref ref-type="bibr" rid="B20">Lazebnik, 2023</xref>). Additionally, efficient control strategies that integrate physics-informed neural networks further reinforce the critical importance of incorporating domain-specific knowledge to enhance both policy performance and system stability (<xref ref-type="bibr" rid="B13">Hu et al., 2024</xref>).</p>
<p>These studies collectively reflect the increasing convergence of reinforcement learning methodologies with domain knowledge integration, safety considerations, and multi-agent coordination principles, which fundamentally underpin the design rationale of our hybrid PPO-GNN framework and its deterministic constraint-validation mechanism. Although these works originate from diverse application domains, their methodological insights provide substantial support for the applicability and potential effectiveness of our approach when applied to complex humanitarian logistics optimization problems.</p>
<p>An additional challenge identified in recent literature concerns the practicality of solution times. Studies demonstrate that fewer than 22% of articles on humanitarian logistics introduce new solution methods designed to deliver results within timeframes practical for field operations (<xref ref-type="bibr" rid="B29">Rodr&#xed;guez-Esp&#xed;ndola et al., 2023</xref>). This gap is particularly problematic in conflict zone logistics, where rapid decision-making can be crucial for operational success. While evolutionary algorithms and Tabu Search represent the most implemented metaheuristics in the field, there remains a significant opportunity to develop specialized solution approaches that effectively balance solution quality with computational efficiency for humanitarian contexts.</p>
<p>Environmental concerns have also emerged as an important consideration in modern humanitarian operations, reflecting broader sustainable development goals (<xref ref-type="bibr" rid="B4">Besiou et al., 2021</xref>). However, our review aligns with previous findings that environmental considerations remain severely underrepresented in humanitarian logistics models, with only a handful of models explicitly incorporating environmental objectives or constraints (<xref ref-type="bibr" rid="B9">Fuli et al., 2020</xref>). This gap represents an important area for future research, as sustainable humanitarian operations become increasingly important in global policy frameworks.</p>
<p>To better illustrate the distinctions between these approaches, we summarize their key advantages and limitations in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Comparison of approaches for humanitarian aid delivery optimization.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method</th>
<th align="left">Strengths</th>
<th align="left">Limitations</th>
<th align="left">Applicability to humanitarian aid</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Classical VRP</td>
<td align="left">Well-studied, efficient heuristics; solid theoretical foundations</td>
<td align="left">Assumes deterministic conditions; limited modeling of uncertainty</td>
<td align="left">Limited use in conflict zones due to dynamic and uncertain environments</td>
</tr>
<tr>
<td align="left">Stochastic and robust VRP</td>
<td align="left">Models uncertainty explicitly; provides robust solutions</td>
<td align="left">High computational complexity; scalability issues for large problems</td>
<td align="left">Challenging for real-time and large-scale humanitarian operations</td>
</tr>
<tr>
<td align="left">Deep reinforcement learning (DRL)</td>
<td align="left">Learns adaptive policies; scalable to complex, dynamic problems</td>
<td align="left">Difficulty handling spatial dependencies; may require large training data</td>
<td align="left">Promising for dynamic routing, but sometimes lacks global network awareness</td>
</tr>
<tr>
<td align="left">Graph neural networks (GNN)</td>
<td align="left">Effectively captures spatial and relational structures; improves feature representation</td>
<td align="left">Requires substantial training data; standalone use limited for sequential decisions</td>
<td align="left">Beneficial when combined with DRL for routing under uncertainty</td>
</tr>
<tr>
<td align="left">Hybrid DRL-GNN approaches</td>
<td align="left">Combines strengths of DRL and GNN; scalable, adaptive, and spatially aware</td>
<td align="left">Relatively new; requires extensive training and validation</td>
<td align="left">High potential for addressing complex humanitarian logistics challenges</td>
</tr>
<tr>
<td align="left">Practitioner-informed models</td>
<td align="left">Aligns modeling objectives with field priorities; increases trust and applicability</td>
<td align="left">Less common in literature; integration with AI methods still emerging</td>
<td align="left">Essential for real-world humanitarian logistics implementation</td>
</tr>
<tr>
<td align="left">Metaheuristics (e.g., Tabu Search, evolutionary algorithms)</td>
<td align="left">Good at providing near-optimal solutions; adaptable to constraints</td>
<td align="left">May require problem-specific tuning; sometimes computationally intensive</td>
<td align="left">Widely used in humanitarian logistics but less explored in hybrid AI contexts</td>
</tr>
<tr>
<td align="left">Environmental-aware models</td>
<td align="left">Addresses sustainability concerns; aligns with global development goals</td>
<td align="left">Underrepresented; lack of integration in most humanitarian models</td>
<td align="left">Important for future-proofing humanitarian logistics planning</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Our work systematically addresses these identified gaps by formulating a rigorous mathematical model for the humanitarian aid delivery problem under uncertainty and developing an innovative hybrid DRL-GNN framework that incorporates practitioner perspectives. This framework strategically combines the strengths of DRL for sequential decision-making and GNNs for capturing spatial and relational structures, enabling efficient and scalable solutions for large-scale humanitarian operations in conflict zones. By integrating chance-constrained formulations and recourse strategies into our framework, we ensure robustness and adaptability in the face of uncertainty. Our approach builds upon recent advances in hybrid AI methods while specifically addressing the unique challenges of humanitarian logistics, such as the critical need for real-time adaptability and the handling of high-dimensional state spaces with security constraints. Through this comprehensive work, we aim to bridge the gap between traditional optimization methods and modern AI techniques, providing a holistic solution for humanitarian aid delivery under uncertainty in conflict zones.</p>
</sec>
<sec id="s3">
<title>3 Problem definition and mathematical model</title>
<sec id="s3-1">
<title>3.1 Original stochastic model</title>
<p>The problem of humanitarian aid delivery in conflict zones is inherently uncertain. To address this, we model aid demand and unloading times as random variables, capturing the unpredictable nature of humanitarian operations. A heterogeneous fleet of vehicles must serve aid distribution centers while considering security risks and operational constraints.</p>
<sec id="s3-1-1">
<title>3.1.1 Objective functions</title>
<p>We define two primary objective functions:</p>
<sec id="s3-1-1-1">
<title>3.1.1.1 Total delivery cost <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</title>
<p>In <xref ref-type="disp-formula" rid="e1">Equation 1</xref>
<disp-formula id="e1">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>Where:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the set of aid delivery locations.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the set of vehicles capable of serving location <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the cost per unit time for vehicle <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of deliveries by vehicle <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to location <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf24">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the unit delivery time, defined as:</p>
</list-item>
</list>
<disp-formula id="e2">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is loading time (<xref ref-type="disp-formula" rid="e2">Equation 2</xref>), <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is uncertain unloading time, <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is distance, and <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf29">
<mml:math id="m31">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> are travel times per kilometer.</p>
<p>This objective aims to minimize operational costs, including fuel, time, and risk exposure.</p>
</sec>
<sec id="s3-1-1-2">
<title>3.1.1.2 Security risk and vehicle dispersion<inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</title>
<p>In <xref ref-type="disp-formula" rid="e3">Equation 3</xref>
<disp-formula id="e3">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>Where:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf31">
<mml:math id="m34">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf32">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> equals 1 if vehicle <inline-formula id="inf33">
<mml:math id="m36">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> serves distribution center <inline-formula id="inf34">
<mml:math id="m37">
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf35">
<mml:math id="m38">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf36">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> equals 1 if vehicle <inline-formula id="inf37">
<mml:math id="m40">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is deployed.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf38">
<mml:math id="m41">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf39">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the security risk coefficient.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf40">
<mml:math id="m43">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf41">
<mml:math id="m44">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf42">
<mml:math id="m45">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are weighting parameters.</p>
</list-item>
</list>
</p>
<p>This objective minimizes the number of vehicles per center, reduces total vehicles deployed, and mitigates security risks.</p>
</sec>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Key constraints</title>
<sec id="s3-1-2-1">
<title>3.1.2.1 Assignment and demand satisfaction constraints</title>
<p>Vehicle Time (<xref ref-type="disp-formula" rid="e4">Equation 4</xref>):<disp-formula id="e4">
<mml:math id="m46">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>Demand Satisfaction (<xref ref-type="disp-formula" rid="e5">Equation 5</xref>):<disp-formula id="e5">
<mml:math id="m47">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>Vehicle Capability (<xref ref-type="disp-formula" rid="e6">Equation 6</xref>):<disp-formula id="e6">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mspace width="1em"/>
<mml:mspace width="1em"/>
<mml:mtext>for&#x2009;&#x2009;</mml:mtext>
<mml:mi>k</mml:mi>
<mml:mo>&#x2209;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>Assignment Variables Linking (<xref ref-type="disp-formula" rid="e7">Equation 7</xref>):<disp-formula id="e7">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>Distribution Center Assignment (<xref ref-type="disp-formula" rid="e8">Equation 8</xref>):<disp-formula id="e8">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="1em"/>
<mml:mtext>when&#x2009;&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>Vehicle Limits (<xref ref-type="disp-formula" rid="e9">Equations 9</xref>, <xref ref-type="disp-formula" rid="e10">10</xref>):<disp-formula id="e9">
<mml:math id="m51">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>and<disp-formula id="e10">
<mml:math id="m52">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-1-2-2">
<title>3.1.2.2 Security constraints</title>
<p>Route Security Threshold (<xref ref-type="disp-formula" rid="e11">Equation 11</xref>):<disp-formula id="e11">
<mml:math id="m53">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>Checkpoint Requirements (<xref ref-type="disp-formula" rid="e12">Equation 12</xref>):<disp-formula id="e12">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mspace width="1em"/>
<mml:mtext>for&#x2009;checkpoints&#x2009;</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mtext>&#x2009;on&#x2009;route&#x2009;</mml:mtext>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>
</p>
<p>Time-dependent Risk (<xref ref-type="disp-formula" rid="e13">Equation 13</xref>):<disp-formula id="e13">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="1em"/>
<mml:mtext>during&#x2009;high</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>risk&#x2009;periods</mml:mtext>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-1-2-3">
<title>3.1.2.3 Route construction constraints</title>
<p>Successor and Predecessor (<xref ref-type="disp-formula" rid="e14">Equation 14</xref>):<disp-formula id="e14">
<mml:math id="m56">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="1em"/>
<mml:mtext>and</mml:mtext>
<mml:mspace width="1em"/>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
</p>
<p>Depot Departure/Return (<xref ref-type="disp-formula" rid="e15">Equation 15</xref>):<disp-formula id="e15">
<mml:math id="m57">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mspace width="1em"/>
<mml:mtext>and</mml:mtext>
<mml:mspace width="1em"/>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
</p>
<p>Position Variables (<xref ref-type="disp-formula" rid="e16">Equation 16</xref>):<disp-formula id="e16">
<mml:math id="m58">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mspace width="1em"/>
<mml:mtext>with</mml:mtext>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
<p>Subtour Elimination (<xref ref-type="disp-formula" rid="e17">Equation 17</xref>):<disp-formula id="e17">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
</p>
</sec>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Deterministic equivalent model</title>
<p>To make the stochastic problem computationally solvable, we transform it into a deterministic equivalent using three approaches:</p>
<sec id="s3-2-1">
<title>3.2.1 Chance-constrained transformation</title>
<p>We replace uncertain demand <inline-formula id="inf43">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> with a certainty-equivalent (<xref ref-type="disp-formula" rid="e18">Equation 18</xref>):<disp-formula id="e18">
<mml:math id="m61">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>
</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(18)</label>
</disp-formula>where <inline-formula id="inf44">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the standard normal quantile for confidence level <inline-formula id="inf45">
<mml:math id="m63">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Thus, the constraint<disp-formula id="equ1">
<mml:math id="m64">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>transforms into (<xref ref-type="disp-formula" rid="e19">Equation 19</xref>)<disp-formula id="e19">
<mml:math id="m65">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>
</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(19)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Expected value transformation</title>
<p>We approximate uncertain unloading time using its expected value (<xref ref-type="disp-formula" rid="e20">Equation 20</xref>):<disp-formula id="e20">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(20)</label>
</disp-formula>which simplifies the unit delivery time (<xref ref-type="disp-formula" rid="e21">Equation 21</xref>):<disp-formula id="e21">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(21)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Recourse transformation</title>
<p>To handle potential shortfalls, we introduce recourse variables <inline-formula id="inf46">
<mml:math id="m68">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf47">
<mml:math id="m69">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> that quantify deviations from target demand with penalty cost <inline-formula id="inf48">
<mml:math id="m70">
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="disp-formula" rid="e22">Equation 22</xref>):<disp-formula id="e22">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>recourse</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>q</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(22)</label>
</disp-formula>with constraints (<xref ref-type="disp-formula" rid="e23">Equations 23</xref>, <xref ref-type="disp-formula" rid="e24">24</xref>):<disp-formula id="e23">
<mml:math id="m72">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(23)</label>
</disp-formula>
<disp-formula id="e24">
<mml:math id="m73">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2265;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(24)</label>
</disp-formula>
</p>
</sec>
<sec id="s3-2-4">
<title>3.2.4 Final deterministic model</title>
<p>The complete deterministic model combines:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf49">
<mml:math id="m74">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Minimizing operational costs <inline-formula id="inf50">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf51">
<mml:math id="m76">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Minimizing recourse costs <inline-formula id="inf52">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>recourse</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf53">
<mml:math id="m78">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Minimizing security risk <inline-formula id="inf54">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</p>
<p>Subject to the deterministic equivalents of all constraints. This reformulation ensures computational feasibility while maintaining robustness in humanitarian operations.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Proposed hybrid methodology</title>
<sec id="s4-1">
<title>4.1 Overview of the hybrid approach</title>
<p>This study relied exclusively on published literature and computational modeling, without direct human subject research. Our methodology synthesizes practitioner priorities documented in humanitarian logistics literature (<xref ref-type="bibr" rid="B11">Holgu&#xed;n-Veras et al., 2013</xref>; <xref ref-type="bibr" rid="B29">Rodr&#xed;guez-Esp&#xed;ndola et al., 2023</xref>) with advanced computational techniques to develop a framework that addresses real-world operational challenges while remaining computationally tractable.</p>
<p>Addressing large-scale humanitarian aid routing in conflict-affected regions presents two primary challenges: the exponential growth of the decision space and the stochastic nature of key operational parameters such as aid demand, route accessibility, and security conditions. Traditional Mixed-Integer Linear Programming (MILP) approaches provide rigorous mathematical formulations but become computationally infeasible for large-scale, real-time decision-making in crisis situations. To overcome these limitations, we propose a hybrid methodology that integrates Deep Reinforcement Learning (DRL), Graph Neural Networks (GNNs), and a post-decision validation mechanism to ensure feasibility and efficiency.</p>
<p>Our approach leverages DRL for adaptive decision-making in complex, high-dimensional spaces, utilizes GNNs to model spatial dependencies within the humanitarian aid distribution network, and incorporates a validation step to refine decisions and enforce operational constraints. This combination enables scalable and adaptive aid delivery optimization while ensuring compliance with real-world feasibility requirements. <xref ref-type="fig" rid="F1">Figure 1</xref> provides an overview of the proposed hybrid methodology, illustrating the interplay between DRL-based decision-making, GNN-enhanced state representation, and the validation mechanism for constraint enforcement.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Overview of the hybrid PPO-GNN framework.</p>
</caption>
<graphic xlink:href="ffutr-06-1603726-g001.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a logistics optimization process using graph neural networks. Key sections include Data Preparation, Graph Construction, VRP Environment, PPO-GNN, PPO (MLP), Clarke-Wright method, Evaluation &#x26; Comparison, Output, Scalability Testing, and Key Innovation. Data preparation involves downloading and processing data. Graph construction involves building a network and validating connectivity. The VRP environment details state and action spaces, along with reward functions. Different methods like PPO-GNN and PPO are trained and evaluated. The chart highlights innovations in graph neural networks, focusing on objectives like distance minimization and risk avoidance.</alt-text>
</graphic>
</fig>
<sec id="s4-1-1">
<title>4.1.1 Learning-based route construction</title>
<p>We employ Proximal Policy Optimization (PPO), a state-of-the-art policy gradient algorithm, to train an agent capable of constructing feasible aid delivery routes dynamically. By continuously interacting with a simulated conflict zone distribution environment, the DRL agent learns to assign deliveries to vehicles, sequence stops efficiently and optimize aid distribution. The policy is trained to minimize total operational costs and security risks while adhering to delivery constraints, such as vehicle capacity, aid availability, security thresholds, and time windows.</p>
</sec>
<sec id="s4-1-2">
<title>4.1.2 Graph-based state representation</title>
<p>Humanitarian aid networks in conflict zones exhibit inherent spatial and relational structures that are best modeled as graphs. To capture these dependencies, we employ a GNN module that processes the delivery network as a graph where nodes represent distribution centers and checkpoints, and edges encode road connectivity, travel distances, security risks, and accessibility conditions. The GNN extracts node embeddings that enrich the DRL state space, providing contextual awareness to improve decision-making. This integration enables the PPO agent to anticipate security threats, congestion patterns, and network-wide operational constraints.</p>
</sec>
<sec id="s4-1-3">
<title>4.1.3 Validation and constraint handling</title>
<p>While the PPO-GNN agent learns efficient delivery policies, it may occasionally generate infeasible solutions due to the stochastic nature of training and the complexity of conflict zone logistics. To mitigate this, a validation mechanism is introduced post-decision-making. This step compares generated routes against the deterministic reference model and assesses compliance with real-world constraints, such as security thresholds, capacity limits, and delivery deadlines. When violations occur, the agent is penalized via reward function adjustments, reinforcing constraint adherence over time. Additionally, fine-tuning and re-training strategies are employed for iterative policy refinement.</p>
</sec>
<sec id="s4-1-4">
<title>4.1.4 Practitioner-informed model design</title>
<p>A key enhancement to our methodology is the incorporation of practitioner perspectives. Through review of published studies documenting perspectives of humanitarian logistics experts (<xref ref-type="bibr" rid="B16">Kov&#xe1;cs and Spens, 2007</xref>; <xref ref-type="bibr" rid="B18">Kunz et al., 2017</xref>; <xref ref-type="bibr" rid="B29">Rodr&#xed;guez-Esp&#xed;ndola et al., 2023</xref>), we identified critical operational priorities and constraints that informed our model design. This engagement revealed that reliability of delivery, quality of service, and prioritization of most affected areas are consistently valued above pure cost minimization. These insights directly influenced our reward function design, constraint formulation, and solution validation criteria.</p>
</sec>
<sec id="s4-1-5">
<title>4.1.5 Scalability and adaptability</title>
<p>The hybrid PPO-GNN framework offers a balance between solution quality and computational efficiency. Unlike MILP-based solvers, which become intractable for large-scale conflict zone operations, PPO-GNN generates near-optimal solutions in a fraction of the time. The learned policy generalizes well to varying network sizes, security disruptions, and demand fluctuations, making it suitable for dynamic, real-world humanitarian crisis response.</p>
<p>By integrating reinforcement learning, graph-based representations, post-decision validation, and practitioner insights, our methodology ensures robust, scalable, and feasible aid delivery routing solutions in volatile environments. The following sections provide detailed insights into the architecture, training process, and experimental validation of the proposed approach.</p>
</sec>
</sec>
<sec id="s4-2">
<title>4.2 Deep reinforcement learning for route construction</title>
<p>The dynamic and stochastic nature of humanitarian aid operations in conflict zones requires a decision-making framework capable of efficiently handling real-time uncertainties while constructing optimized delivery routes. To address this challenge, we employ DRL, which enables adaptive learning of optimal routing strategies by interacting with a simulated environment.</p>
<sec id="s4-2-1">
<title>4.2.1 DRL model architecture</title>
<p>Our DRL framework is structured as a Markov Decision Process (MDP) defined by the tuple <inline-formula id="inf55">
<mml:math id="m80">
<mml:mrow>
<mml:mo stretchy="false">&#x27e8;</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x27e9;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, where:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf56">
<mml:math id="m81">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf57">
<mml:math id="m82">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the state space, encoding relevant information such as vehicle locations, aid demands at distribution centers, security conditions, checkpoint statuses, and road accessibility.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf58">
<mml:math id="m83">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf59">
<mml:math id="m84">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> defines the action space, consisting of feasible routing decisions, including vehicle selection, order sequencing, and checkpoint traversal strategies.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf60">
<mml:math id="m85">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf61">
<mml:math id="m86">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the transition probability, which models the environment dynamics after executing action <inline-formula id="inf62">
<mml:math id="m87">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in state <inline-formula id="inf63">
<mml:math id="m88">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf64">
<mml:math id="m89">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf65">
<mml:math id="m90">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the reward function, designed to optimize humanitarian aid distribution efficiency while penalizing security risks and infeasible actions.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf66">
<mml:math id="m91">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf67">
<mml:math id="m92">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the discount factor, which balances immediate versus long-term rewards.</p>
</list-item>
</list>
</p>
<p>For policy optimization, we adopt PPO, a robust and sample-efficient policy gradient algorithm that ensures stable training and effective exploration-exploitation trade-offs. PPO is particularly well-suited for our problem as it efficiently handles large-scale decision spaces and dynamically changing constraints, both of which are crucial in conflict zone humanitarian logistics.</p>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Integration of graph neural networks with PPO</title>
<p>While PPO provides a strong foundation for policy optimization, it struggles to capture the complex spatial dependencies inherent in humanitarian aid networks in conflict zones. To address this limitation, we integrate a GNN module into the PPO framework. This integration enables the agent to leverage the relational structure of the delivery network, improving its ability to generalize and adapt to dynamic environments with changing security conditions.</p>
<sec id="s4-2-2-1">
<title>4.2.2.1 Graph representation of the delivery network</title>
<p>The humanitarian aid delivery network is modeled as graph <inline-formula id="inf68">
<mml:math id="m93">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf69">
<mml:math id="m94">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf70">
<mml:math id="m95">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents nodes, including aid distribution centers, checkpoints, and depots, each characterized by attributes such as aid demand, security risk level, accessibility status, and capacity constraints.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf71">
<mml:math id="m96">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf72">
<mml:math id="m97">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents edges, capturing connectivity between locations and associated travel times, distances, security conditions, and accessibility.</p>
</list-item>
</list>
</p>
<p>The GNN processes this graph to generate node embeddings <inline-formula id="inf73">
<mml:math id="m98">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which encode spatial and operational characteristics. These embeddings are then integrated into the state representation, enriching the agent&#x2019;s understanding of the conflict zone environment.</p>
</sec>
<sec id="s4-2-2-2">
<title>4.2.2.2 Integration of GNN into PPO</title>
<p>The GNN-enhanced PPO framework operates as follows:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf74">
<mml:math id="m99">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Graph Embedding Generation: The GNN computes node embeddings <inline-formula id="inf75">
<mml:math id="m100">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> by aggregating information from neighboring nodes.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf76">
<mml:math id="m101">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> State Representation Augmentation: The embeddings <inline-formula id="inf77">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are concatenated with traditional state features (e.g., vehicle status, pending deliveries, security alerts) to form an enriched state representation <inline-formula id="inf78">
<mml:math id="m103">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf79">
<mml:math id="m104">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Policy and Value Function Enhancement: The augmented state is passed to the PPO policy and value networks, enabling the agent to incorporate graph-based insights into its decision-making process.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf80">
<mml:math id="m105">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Action Selection: The PPO policy selects optimal routing and vehicle assignment actions based on enhanced state information.</p>
</list-item>
</list>
</p>
<p>This integration allows the agent to make globally optimized decisions by leveraging both local and global network structures. <xref ref-type="fig" rid="F2">Figure 2</xref> illustrates the architecture of the PPO-GNN framework, highlighting the interaction between the GNN and PPO components.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>PPO-GNN architecture.</p>
</caption>
<graphic xlink:href="ffutr-06-1603726-g002.tif">
<alt-text content-type="machine-generated">Graph illustrating a reinforcement learning framework. The left part shows graph representation with nodes and edges, representing distribution centers and roads. GNN layers extract node embeddings, linking to the reinforcement learning component. This component includes a policy network for vehicle assignment and route selection, a value network for state value estimation, and a PPO update system with objectives and loss functions. Actions like routing decisions are indicated.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s4-2-3">
<title>4.2.3 Reward function and optimization criteria</title>
<p>The implemented reward function (<xref ref-type="disp-formula" rid="e25">Equation 25</xref>) for the PPO and PPO-GNN agents incorporates four weighted components aligned with key humanitarian logistics priorities identified through practitioner input and literature (<xref ref-type="bibr" rid="B11">Holgu&#xed;n-Veras et al., 2013</xref>; <xref ref-type="bibr" rid="B33">Wassenhove, 2006</xref>):<disp-formula id="e25">
<mml:math id="m106">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>distance</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>risk</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>unmet</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>violation</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(25)</label>
</disp-formula>where:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf81">
<mml:math id="m107">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf82">
<mml:math id="m108">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>distance</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalizes total travel distance (operational cost).</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf83">
<mml:math id="m109">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf84">
<mml:math id="m110">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>risk</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalizes exposure to security risks in conflict zones.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf85">
<mml:math id="m111">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf86">
<mml:math id="m112">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>unmet</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalizes unmet humanitarian aid demands.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf87">
<mml:math id="m113">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf88">
<mml:math id="m114">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>violation</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> penalizes vehicle capacity constraint violations.</p>
</list-item>
</list>
</p>
<p>The corresponding weights are set as:<disp-formula id="equ2">
<mml:math id="m115">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.0</mml:mn>
<mml:mspace width="1em"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>distance</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>100.0</mml:mn>
<mml:mspace width="1em"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>risk</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>10.0</mml:mn>
<mml:mspace width="1em"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>unmet&#x2009;demand</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>50.0</mml:mn>
<mml:mspace width="1em"/>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mtext>capacity&#x2009;violation</mml:mtext>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>This formulation pragmatically balances operational costs and security concerns, emphasizing risk mitigation while ensuring aid delivery effectiveness. Other relevant criteria such as delay penalties, delivery reliability, and aid quality, identified as important by field experts, are considered for future extensions but are not explicitly modeled in this version.</p>
<p>These coefficients were calibrated in consultation with humanitarian practitioners to reflect real-world priorities in conflict-affected logistics operations, enabling the reinforcement learning agents to optimize routing strategies accordingly.</p>
</sec>
<sec id="s4-2-4">
<title>4.2.4 Training process and policy learning</title>
<p>The DRL agent interacts with a simulated humanitarian aid delivery environment modeled after conflict-affected regions, collecting experiences and refining its policy through iterative updates. The training process follows a policy gradient approach, where the policy <inline-formula id="inf89">
<mml:math id="m116">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is updated to maximize the expected return (<xref ref-type="disp-formula" rid="e26">Equation 26</xref>):<disp-formula id="e26">
<mml:math id="m117">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>J</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(26)</label>
</disp-formula>Here, <inline-formula id="inf90">
<mml:math id="m118">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents the advantage function, which estimates the relative value of action <inline-formula id="inf91">
<mml:math id="m119">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in state <inline-formula id="inf92">
<mml:math id="m120">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> by normalizing expected rewards, thereby reducing variance and improving training stability.</p>
<p>To further stabilize learning, we simultaneously optimize a value function <inline-formula id="inf93">
<mml:math id="m121">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> using mean squared error loss (<xref ref-type="disp-formula" rid="e27">Equation 27</xref>):<disp-formula id="e27">
<mml:math id="m122">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(27)</label>
</disp-formula>
</p>
<p>The training process employs mini-batch gradient updates and adaptive learning rate scheduling to enhance efficiency and convergence in the context of conflict zone humanitarian operations.</p>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 Validation and constraint handling</title>
<p>To ensure that the solutions generated by the PPO-GNN framework remain feasible under conflict zone operational constraints, a post-training validation process is integrated. This process compares the learned policies with the deterministic reference model and penalizes constraint violations, ensuring adherence to real-world feasibility conditions.</p>
<p>The validation step identifies infeasible actions, such as routing through high-risk areas, exceeding vehicle capacity, or violating humanitarian access protocols, and introduces adaptive penalties in the reward function. These penalties discourage infeasible solutions by assigning higher negative rewards to constraint violations, thereby steering the DRL agent toward more compliant policies.</p>
<p>Based on practitioner feedback, we developed a two-tier validation approach:<list list-type="simple">
<list-item>
<p>1. Critical Constraint Validation: Enforces non-negotiable constraints such as security thresholds, access permissions, and minimum aid requirements. Violations of these constraints trigger immediate correction or solution rejection.</p>
</list-item>
<list-item>
<p>2. Flexible Constraint Validation: Handles soft constraints such as preferred delivery times and vehicle utilization targets. Violations of these constraints incur proportional penalties but allow solutions to be accepted with minor deviations when necessary.</p>
</list-item>
</list>
</p>
<p>This tiered approach aligns with real operational practices in humanitarian logistics, where field practitioners often must balance ideal conditions with practical realities.</p>
</sec>
<sec id="s4-4">
<title>4.4 Benchmarking and evaluation framework</title>
<p>To provide a comprehensive assessment of our PPO-GNN framework, we established a benchmarking and evaluation protocol informed by both academic standards and practitioner requirements. The evaluation metrics include:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf94">
<mml:math id="m123">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Cost Efficiency: Total operational costs including fuel, personnel, and vehicle usage.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf95">
<mml:math id="m124">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Security Risk Exposure: Cumulative security risk across all routes.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf96">
<mml:math id="m125">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Demand Satisfaction: Percentage of aid demand fulfilled across distribution centers.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf97">
<mml:math id="m126">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Time Efficiency: Total delivery time and adherence to time windows.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf98">
<mml:math id="m127">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Reliability: Consistency of service across multiple simulation runs.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf99">
<mml:math id="m128">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Quality of Service: Appropriate matching aid types to specific needs.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf100">
<mml:math id="m129">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Adaptability: Performance under varying security conditions and demand patterns.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf101">
<mml:math id="m130">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Computational Efficiency: Solution time and resource requirements.</p>
</list-item>
</list>
</p>
<p>These metrics allow for a multidimensional comparison with baseline approaches, reflecting both operational efficiency and humanitarian effectiveness. The next section presents the experimental results of this evaluation across various conflict zone scenarios.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Experimental evaluation</title>
<p>We evaluate the performance of the PPO-GNN algorithm on humanitarian aid delivery problems in conflict-affected regions, comparing it against two baselines: (i) a classical PPO agent without graph neural network augmentation, to isolate the impact of graph representation learning; and (ii) the Clarke-Wright savings heuristic, a well-established non-learning benchmark. Our evaluation metrics focus on solution quality, exposure to security risks, computational efficiency, and robustness to stochastic variations in demand, route accessibility, and security conditions.</p>
<sec id="s5-1">
<title>5.1 Experimental setup</title>
<sec id="s5-1-1">
<title>5.1.1 Network configurations</title>
<p>To rigorously assess the effectiveness and scalability of our method under realistic operational conditions, we generate three representative benchmark instances by extracting subgraphs from a high-resolution, georeferenced road and logistics network of Afghanistan. This approach ensures that each benchmark instance faithfully captures the spatial, topological, and operational complexities characteristic of real humanitarian logistics, while allowing controlled scalability analysis.</p>
<p>The synthetic data generation framework is parameterized to produce three operational scales:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf102">
<mml:math id="m131">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Small-scale network (<inline-formula id="inf103">
<mml:math id="m132">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mn>50</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> nodes, <inline-formula id="inf104">
<mml:math id="m133">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mn>120</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> edges): Simulates localized humanitarian operations, such as district-level aid delivery in confined conflict zones.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf105">
<mml:math id="m134">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Medium-scale network (<inline-formula id="inf106">
<mml:math id="m135">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mn>150</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> nodes, <inline-formula id="inf107">
<mml:math id="m136">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mn>400</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> edges): Models regional humanitarian responses spanning several districts within a conflict zone, typical of mid-sized crisis interventions.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf108">
<mml:math id="m137">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Large-scale network (<inline-formula id="inf109">
<mml:math id="m138">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mn>500</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> nodes, <inline-formula id="inf110">
<mml:math id="m139">
<mml:mrow>
<mml:mo>&#x223c;</mml:mo>
<mml:mn>2000</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> edges): Represents large-scale national or multi-regional humanitarian operations often managed by international agencies.</p>
</list-item>
</list>
</p>
<p>It is important to note that the actual subgraphs extracted from the Afghanistan data used in our experiments may differ in size due to data availability and network characteristics. For instance, the large-scale instance employed in our experiments contains 81 nodes and 89 edges, reflecting the true connectivity and spatial distribution of the underlying infrastructure.</p>
<p>Nodes correspond to geographical coordinates of real or plausible distribution centers identified through humanitarian logistics datasets and field reports. Edges represent actual road segments, with distances computed via the WGS84 geodesic formula to ensure geographic accuracy. Edge risk levels are assessed based on proximity to recent conflict events, leveraging the UCDP Georeferenced Event Dataset (GED) and established spatial risk assessment methodologies (<xref ref-type="bibr" rid="B29">Rodr&#xed;guez-Esp&#xed;ndola et al., 2023</xref>).</p>
<p>Subgraphs are extracted by selecting geographically coherent clusters that maintain spatial contiguity, connectivity, and operational features typical of humanitarian logistics networks. Node demands are modeled as random variables <inline-formula id="inf111">
<mml:math id="m140">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> based on historical or simulated aid distribution data, thereby capturing the stochastic nature of humanitarian needs. Security risk levels and route accessibility attributes are directly inherited from the source network data.</p>
</sec>
<sec id="s5-1-2">
<title>5.1.2 Practitioner-informed experimental design</title>
<p>Our experimental design is grounded in best practices derived from the literature and informed by humanitarian logistics experts with direct field experience in conflict-affected regions (<xref ref-type="bibr" rid="B31">Tomasini and Wassenhove, 2009</xref>; <xref ref-type="bibr" rid="B26">Pedraza-Martinez and Wassenhove, 2013</xref>; <xref ref-type="bibr" rid="B16">Kov&#xe1;cs and Spens, 2007</xref>). This ensures the computational experiments realistically reflect operational realities and practitioner priorities.</p>
<p>Key design considerations include:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf112">
<mml:math id="m141">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Realistic network structures and constraints: Network topologies and operational limits are based on documented humanitarian logistics configurations, ensuring fidelity to actual field conditions encountered in conflict zones.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf113">
<mml:math id="m142">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Security risk modeling: We incorporate security risk assessments aligned with established humanitarian security frameworks, integrating spatial proximity to recent conflict events and dynamic accessibility conditions.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf114">
<mml:math id="m143">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Demand uncertainty: Node demand distributions are parameterized using historical aid delivery data and simulated stochastic variations, capturing the volatile and uncertain nature of humanitarian needs.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf115">
<mml:math id="m144">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Performance metrics: Evaluation criteria are selected based on practitioner-focused studies (<xref ref-type="bibr" rid="B29">Rodr&#xed;guez-Esp&#xed;ndola et al., 2023</xref>), emphasizing solution quality, security exposure, unmet demand, and operational feasibility.</p>
</list-item>
</list>
</p>
<p>By embedding empirical knowledge and leveraging authentic network and security data, our experimental framework bridges the gap between computational methods and operational applicability, ensuring that the results are both scientifically rigorous and practically relevant for humanitarian logistics decision-makers.</p>
</sec>
<sec id="s5-1-3">
<title>5.1.3 Implementation details</title>
<p>This section details the practical implementation of our adaptive vehicle routing methodology for humanitarian aid distribution in conflict-affected regions, focusing on the Afghanistan use case.</p>
<sec id="s5-1-3-1">
<title>5.1.3.1 Data sources and preparation</title>
<p>
<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf116">
<mml:math id="m145">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Road Network Data: We use the <monospace>hotosm_afg_roads_lines</monospace> dataset, an OpenStreetMap-derived shapefile that includes primary, secondary, tertiary, and unclassified roads across Afghanistan, filtered to remove unpaved or low-quality segments. This forms the spatial backbone for our routing graph.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf117">
<mml:math id="m146">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Conflict Event Data: The <monospace>GEDEvent_v23_1</monospace> and <monospace>ged_afg</monospace> datasets originate from the Uppsala Conflict Data Program (UCDP) Georeferenced Event Dataset (GED). They provide geolocated conflict incidents, enabling us to spatially estimate risk exposure levels on road segments by buffering edges and counting proximate conflict events normalized by buffer area.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf118">
<mml:math id="m147">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Demand Data: The dataset contains humanitarian aid demand estimates per distribution center node, modeled based on historical aid distribution records and domain expert insights. Each node is assigned stochastic demand parameters (mean <inline-formula id="inf119">
<mml:math id="m148">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and standard deviation <inline-formula id="inf120">
<mml:math id="m149">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>) reflecting the uncertain nature of humanitarian needs in these regions.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s5-1-3-2">
<title>5.1.3.2 Graph generation</title>
<p>
<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf121">
<mml:math id="m150">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Using <monospace>networkx</monospace> and <monospace>geopandas</monospace>, the road network is represented as a weighted undirected graph <inline-formula id="inf122">
<mml:math id="m151">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>V</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where nodes <inline-formula id="inf123">
<mml:math id="m152">
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> correspond to geographic coordinates (longitude-latitude tuples) and edges <inline-formula id="inf124">
<mml:math id="m153">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent road segments. Edge weights reflect geodesic distances computed via the <monospace>geopy</monospace> library. Risk scores for edges are derived by buffering each road segment and counting overlapping conflict events from the GED dataset, normalized by the buffer area to quantify security exposure.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf125">
<mml:math id="m154">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> To create scalable problem instances, a connected subgraph is extracted by selecting a central node with high degree centrality and performing breadth-first search expansions until the target number of nodes and edges is met. Connectivity is maintained by adding bridging edges as necessary. This approach balances computational feasibility with realistic operational scenarios.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s5-1-3-3">
<title>5.1.3.3 Algorithms and training</title>
<p>
<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf126">
<mml:math id="m155">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Heuristic Baseline: The Clarke-Wright savings heuristic is implemented with Google OR-Tools. The graph&#x2019;s distance matrix is computed using all-pairs shortest paths via Dijkstra&#x2019;s algorithm. Vehicle capacity and demand constraints are integrated, and the heuristic outputs solution metrics and route sequences.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf127">
<mml:math id="m156">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Deep Reinforcement Learning Models: Two PPO-based agents are trained:- A classical PPO agent with a multilayer perceptron (MLP) policy.- A PPO agent enhanced with a graph convolutional network (GCN) feature extractor to leverage spatial and topological information.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf128">
<mml:math id="m157">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Environment: The routing problem is modeled as a custom OpenAI Gym environment (<monospace>HumanitarianVRP</monospace>), encapsulating stochastic node demands per episode. Rewards combine weighted costs of distance traveled, risk exposure, unmet demand, and vehicle capacity violations.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf129">
<mml:math id="m158">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Training Parameters and Computational Details: Both PPO agents are trained for 5,000 episodes using Stable Baselines3, with the following key hyperparameters:- Learning rate: <inline-formula id="inf130">
<mml:math id="m159">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>- Discount factor: <inline-formula id="inf131">
<mml:math id="m160">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.99</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>- GAE lambda: 0.95- Batch size: 64- Number of epochs per update: 4- PPO clip range: 0.2</p>
</list-item>
</list>
</p>
<p>The total number of trainable parameters is:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf132">
<mml:math id="m161">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Approximately 75,986 for the PPO (MLP) model</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf133">
<mml:math id="m162">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Approximately 47,250 for the PPO-GNN model with GCN extractor</p>
</list-item>
</list>
</p>
<p>Training duration per run is approximately 10&#xa0;s on CPU for 5,000 episodes.</p>
<p>The environment parameters (coefficients in the reward function) are fixed as follows:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf134">
<mml:math id="m163">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf135">
<mml:math id="m164">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (distance weight)</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf136">
<mml:math id="m165">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf137">
<mml:math id="m166">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>100.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (risk weight)</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf138">
<mml:math id="m167">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf139">
<mml:math id="m168">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>10.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (unmet demand weight)</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf140">
<mml:math id="m169">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf141">
<mml:math id="m170">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>50.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (capacity violation weight)</p>
</list-item>
</list>
</p>
<p>The maximum number of steps per episode is set to 162, and the vehicle capacity is fixed at 1,000 units.</p>
</sec>
<sec id="s5-1-3-4">
<title>5.1.3.4 Route extraction</title>
<p>Post-training, policies are evaluated over multiple test episodes with fixed random seeds to ensure robustness. The resulting sequences of visited nodes (routes) are saved for detailed analysis and visualization.</p>
</sec>
<sec id="s5-1-3-5">
<title>5.1.3.5 Software and reproducibility</title>
<p>
<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf142">
<mml:math id="m171">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Codebase: The implementation is modular, comprising distinct scripts for data preprocessing, model training, heuristic evaluation, and visualization.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf143">
<mml:math id="m172">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> File Organization: Input data (graphs, demand, conflict events) and output logs are systematically organized in dedicated directories (e.g., <monospace>data/raw</monospace>, <monospace>data/proc</monospace>, <monospace>results/logs</monospace>) to promote reproducibility.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf144">
<mml:math id="m173">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Visualization: Route visualizations overlay optimal paths on Afghanistan&#x2019;s road network shapefile using <monospace>geopandas</monospace> and <monospace>matplotlib</monospace>, featuring automatic zoom and informative legends to support method comparisons across scales.</p>
</list-item>
</list>
</p>
</sec>
<sec id="s5-1-3-6">
<title>5.1.3.6 Code availability</title>
<p>The complete source code and datasets supporting this research are publicly available in the PPO-GNN-humanitarian GitHub repository under the permissive MIT license. This repository enables full reproducibility of all experiments and results presented in this paper: <ext-link ext-link-type="uri" xlink:href="https://github.com/ARGOUBI25/PPO-GNN-humanitarian">https://github.com/ARGOUBI25/PPO-GNN-humanitarian</ext-link>
</p>
<p>The repository includes:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf145">
<mml:math id="m174">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Scripts for data preprocessing, including graph construction and demand modeling.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf146">
<mml:math id="m175">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Implementations of the heuristic and reinforcement learning algorithms (PPO and PPO-GNN).</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf147">
<mml:math id="m176">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Training and evaluation workflows.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf148">
<mml:math id="m177">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Visualization tools for route plotting and analysis.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf149">
<mml:math id="m178">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Configuration files and detailed instructions for replicating experiments at multiple scales (small, medium, large).</p>
</list-item>
</list>
</p>
<p>Users are encouraged to consult the README and accompanying documentation for setup and usage details.</p>
</sec>
</sec>
<sec id="s5-1-4">
<title>5.1.4 Practical implementation and adoption considerations</title>
<p>Beyond the rigorous computational experiments presented, practical deployment of the PPO-GNN framework in humanitarian operations requires careful consideration of scalability, real-time adaptability, and integration with existing decision-making processes. The method&#x2019;s ability to handle large-scale networks with stochastic demand and dynamic security risks positions it well for supporting field logistics under volatile conditions. To facilitate adoption, ongoing collaborations with humanitarian organizations are planned to conduct real-world validation and co-design workflows that align with operational constraints and practitioner needs. Such partnerships will enable iterative refinement of the framework based on direct user feedback, ensuring that the algorithmic advances translate into actionable, trustworthy tools. Addressing challenges related to resource constraints, interpretability, and training infrastructure will be key to realizing the framework&#x2019;s potential as a decision support system in complex conflict-affected environments.</p>
</sec>
</sec>
<sec id="s5-2">
<title>5.2 Performance comparison</title>
<p>
<xref ref-type="table" rid="T2">Table 2</xref> presents the performance comparison of the PPO-GNN agent against classical PPO and the Clarke-Wright heuristic, including mean values and standard deviations calculated over multiple independent runs with different random seeds. Reporting these statistics enables assessment of variability and reliability of the methods.<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf150">
<mml:math id="m179">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Total Cost: PPO-GNN achieves the lowest mean total cost ($12,800 <inline-formula id="inf151">
<mml:math id="m180">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 300), representing a statistically significant reduction compared to classical PPO ($13,900 <inline-formula id="inf152">
<mml:math id="m181">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 450) and Clarke-Wright ($14,300 <inline-formula id="inf153">
<mml:math id="m182">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 100). This cost saving is attributed to the GNN&#x2019;s ability to exploit spatial and security-related features for route optimization.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf154">
<mml:math id="m183">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Security Risk Exposure: PPO-GNN demonstrates superior risk mitigation with a mean exposure of 325.6 (<inline-formula id="inf155">
<mml:math id="m184">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 15.4), substantially lower than classical PPO and Clarke-Wright. The standard deviations indicate consistent performance across runs.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf156">
<mml:math id="m185">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Unmet Demand: The PPO-GNN agent achieves the lowest average unmet demand at 2.0% (&#xb1;0.5), outperforming classical PPO and Clarke-Wright, highlighting its robustness under stochastic demand and security conditions.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf157">
<mml:math id="m186">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Solve Time: While the Clarke-Wright heuristic remains the fastest (2.4 &#xb1; 0.2&#xa0;s), PPO-GNN (18.3 &#xb1; 1.5&#xa0;s) and classical PPO (15.7 &#xb1; 1.3&#xa0;s) incur higher computational costs, justified by their improved solution quality in humanitarian contexts.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf158">
<mml:math id="m187">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Constraint Violations: PPO-GNN maintains the lowest rate of constraint violations (1.0% &#xb1; 0.3), evidencing effective adherence to operational limits.</p>
</list-item>
</list>
</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Performance on synthetic datasets (all reproducible via GitHub repository).</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Metric</th>
<th align="center">PPO-GNN (mean <inline-formula id="inf159">
<mml:math id="m188">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> std)</th>
<th align="center">Classical PPO (Mean <inline-formula id="inf160">
<mml:math id="m189">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> std)</th>
<th align="center">Clarke-Wright <inline-formula id="inf161">
<mml:math id="m190">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> PPO-GNN</th>
<th align="center">
<inline-formula id="inf162">
<mml:math id="m191">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> PPO-GNN vs. PPO (%)</th>
<th align="center">
<inline-formula id="inf163">
<mml:math id="m192">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> PPO-GNN vs. CW (%)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Total cost (USD)</td>
<td align="center">
<inline-formula id="inf164">
<mml:math id="m193">
<mml:mrow>
<mml:mn>12,800</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>30</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf165">
<mml:math id="m194">
<mml:mrow>
<mml:mn>13,900</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>450</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf166">
<mml:math id="m195">
<mml:mrow>
<mml:mn>14,300</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>10</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">&#x2212;7.91%</td>
<td align="center">&#x2212;10.49%</td>
</tr>
<tr>
<td align="left">Security risk exposure</td>
<td align="center">
<inline-formula id="inf167">
<mml:math id="m196">
<mml:mrow>
<mml:mn>325.6</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>15</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf168">
<mml:math id="m197">
<mml:mrow>
<mml:mn>383.9</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>20.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf169">
<mml:math id="m198">
<mml:mrow>
<mml:mn>426.8</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>18</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">&#x2212;15.17%</td>
<td align="center">&#x2212;23.73%</td>
</tr>
<tr>
<td align="left">Unmet demand (%)</td>
<td align="center">
<inline-formula id="inf170">
<mml:math id="m199">
<mml:mrow>
<mml:mn>2.0</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf171">
<mml:math id="m200">
<mml:mrow>
<mml:mn>8.0</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf172">
<mml:math id="m201">
<mml:mrow>
<mml:mn>5.0</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>8</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">&#x2212;75.00%</td>
<td align="center">&#x2212;60.00%</td>
</tr>
<tr>
<td align="left">Solve time (s)</td>
<td align="center">
<inline-formula id="inf173">
<mml:math id="m202">
<mml:mrow>
<mml:mn>18.3</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>1.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf174">
<mml:math id="m203">
<mml:mrow>
<mml:mn>15.7</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>1.3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf175">
<mml:math id="m204">
<mml:mrow>
<mml:mn>2.4</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>0.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">16.56%</td>
<td align="center">662.50%</td>
</tr>
<tr>
<td align="left">Constraint violations (%)</td>
<td align="center">
<inline-formula id="inf176">
<mml:math id="m205">
<mml:mrow>
<mml:mn>1.0</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf177">
<mml:math id="m206">
<mml:mrow>
<mml:mn>6.0</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>1.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">
<inline-formula id="inf178">
<mml:math id="m207">
<mml:mrow>
<mml:mn>3.0</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>.</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">&#x2212;83.33%</td>
<td align="center">&#x2212;66.67%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Statistical significance tests (Wilcoxon signed-rank) conducted between PPO-GNN and classical PPO confirm that observed differences in cost, risk, and unmet demand metrics are statistically significant <inline-formula id="inf179">
<mml:math id="m208">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.05</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, supporting the robustness of our approach. Full details of these tests and variance analyses are provided in the full details of these tests and variance analyses, along with complete code for reproduction, are available in the GitHub repository: <ext-link ext-link-type="uri" xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="https://github.com/ARGOUBI25/PPO-GNN-humanitarian">https://github.com/ARGOUBI25/PPO-GNN-humanitarian</ext-link>.</p>
</sec>
<sec id="s5-3">
<title>5.3 Scalability analysis</title>
<p>To assess the scalability of our approach, we evaluated performance metrics across different network sizes. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates how solution quality and computational efficiency scale with increasing problem size.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Performance across network sizes.</p>
</caption>
<graphic xlink:href="ffutr-06-1603726-g003.tif">
<alt-text content-type="machine-generated">Four bar charts compare decision-making algorithms across three network sizes: small (10 nodes), medium (50 nodes), and large (100 nodes). Each chart represents different metrics: total cost, security risk exposure, unmet demand, and solve time. Three algorithms&#x2014;PPO-GNN, Classical PPO, and Clarke-Wright&#x2014;are indicated in blue, orange, and green, respectively. The legend is at the bottom. Total cost and solve time charts show minor variations among algorithms, while security risk and unmet demand exhibit notable differences, especially in larger networks.</alt-text>
</graphic>
</fig>
<p>For small networks (10 nodes), all methods perform reasonably well, with PPO-GNN showing modest improvements in solution quality. However, as network size increases, the benefits of our approach become more pronounced. In medium networks (50 nodes), PPO-GNN demonstrates a 12% cost reduction over classical PPO and a 16% reduction over Clarke-Wright, while maintaining acceptable solution times.</p>
<p>The most significant advantages appear in large networks (100 nodes), where PPO-GNN achieves a 21% cost reduction compared to classical PPO and a 27% reduction compared to Clarke-Wright. While solution times increase for all methods in larger networks, PPO-GNN&#x2019;s computation time grows at a more manageable rate than might be expected for such complex optimization problems, demonstrating the scalability of our approach.</p>
</sec>
<sec id="s5-4">
<title>5.4 Robustness to stochastic variations</title>
<p>A critical aspect of humanitarian aid delivery in conflict zones is robustness to unexpected variations in demand, security conditions, and route accessibility. We evaluated this robustness through simulation experiments where these parameters were subject to random fluctuations beyond their expected distributions.</p>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> depicts the performance degradation of each method under increasing levels of stochasticity. PPO-GNN demonstrates superior robustness, with only a 14% performance degradation under severe stochastic conditions, compared to 29% for classical PPO and 41% for Clarke-Wright. This enhanced robustness can be attributed to the GNN&#x2019;s ability to encode spatial relationships that remain relatively stable even as individual node and edge attributes fluctuate.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Robustness to stochastic variations.</p>
</caption>
<graphic xlink:href="ffutr-06-1603726-g004.tif">
<alt-text content-type="machine-generated">Line graph showing performance degradation percentage on the y-axis against stochasticity level on the x-axis. Three lines represent different methods: PPO-GNN (blue circles), Classical PPO (orange squares), and Clarke-Wright (green triangles). The graph indicates an increase in performance degradation with higher stochasticity levels, with Clarke-Wright reaching 40%, Classical PPO 29%, and PPO-GNN 14% at very high levels. A legend explains stochasticity includes demand fluctuations, security condition changes, and route accessibility variations.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s5-5">
<title>5.5 Practitioner-validated metrics</title>
<p>Based on priorities identified in humanitarian logistics practitioner literature (<xref ref-type="bibr" rid="B32">Vega and Roussat, 2015</xref>; <xref ref-type="bibr" rid="B10">Galindo and Batta, 2013</xref>), we developed and evaluated additional metrics that align with field priorities:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf180">
<mml:math id="m209">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Service Reliability: Measured as the consistency of delivery schedules across multiple simulation runs with varying conditions. PPO-GNN achieved 89% reliability, compared to 72% for classical PPO and 65% for Clarke-Wright.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf181">
<mml:math id="m210">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Aid Quality Matching: Evaluated how well the algorithm matched specific aid types to location needs. PPO-GNN correctly matched aid types in 94% of cases, compared to 81% for classical PPO and 76% for Clarke-Wright.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf182">
<mml:math id="m211">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Operational Adaptability: Assessed through simulated disruption scenarios where certain routes became suddenly inaccessible. PPO-GNN successfully rerouted 87% of affected deliveries within acceptable timeframes, compared to 65% for classical PPO and 52% for Clarke-Wright.</p>
</list-item>
</list>
</p>
<p>These results highlight that beyond traditional optimization metrics, our approach also excels in dimensions highly valued by humanitarian practitioners, reinforcing its potential for real-world implementation.</p>
</sec>
<sec id="s5-6">
<title>5.6 Visual analysis of routing solutions</title>
<p>To further elucidate the qualitative differences between the routing solutions produced by the evaluated methods, <xref ref-type="fig" rid="F5">Figure 5</xref> presents side-by-side visualizations of routes generated by the Clarke-Wright heuristic, classical PPO, and the proposed PPO-GNN approach on a large-scale real-world road network.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Visualization of routing solutions.</p>
</caption>
<graphic xlink:href="ffutr-06-1603726-g005.tif">
<alt-text content-type="machine-generated">Three side-by-side maps compare route optimization methods over the same geographic area. The left map, labeled &#x22;Clarke-Wright Heuristic,&#x22; shows a network of red nodes and lines. The center map, &#x22;PPO (Deep RL),&#x22; features orange nodes and dashed lines. The right map, &#x22;PPO-GNN (Our Approach),&#x22; illustrates green nodes and lines, indicating a different route distribution.</alt-text>
</graphic>
</fig>
<p>The figure distinctly highlights the comparative performance:<list list-type="simple">
<list-item>
<p>
<inline-formula id="inf183">
<mml:math id="m212">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Clarke-Wright heuristic routes (left panel, orange) exhibit fragmented and less coherent paths, with many short detours and overlapping segments. The routes lack global optimization awareness and sometimes revisit nodes inefficiently, reflecting the heuristic&#x2019;s limited consideration of complex spatial and security factors.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf184">
<mml:math id="m213">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> Classical PPO routes (center panel, blue) demonstrate more consolidated and logical paths than Clarke-Wright, with fewer unnecessary detours and better continuity between nodes. However, some inefficiencies remain, including occasional route overlap and suboptimal navigation around risk-prone areas.</p>
</list-item>
<list-item>
<p>
<inline-formula id="inf185">
<mml:math id="m214">
<mml:mrow>
<mml:mo>&#x2022;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> PPO-GNN routes (right panel, green), representing our proposed hybrid approach, reveal clear improvements in route quality. These routes are smoother, better structured, and more direct, effectively minimizing route overlap and unnecessary traversal. The integration of graph neural networks enables encoding of spatial dependencies and security risk profiles, allowing the agent to generate safer, more efficient routing solutions.</p>
</list-item>
</list>
</p>
<p>All panels display the same set of centers (red dots) and start/end points (square and circle markers), ensuring comparability. The PPO-GNN routes also show the greatest coherence in path progression, indicating superior handling of operational constraints and stochastic demand.</p>
<p>This visual evidence complements the quantitative performance metrics, providing tangible proof of PPO-GNN&#x2019;s enhanced routing strategy in a challenging real-world context. Such qualitative insights are crucial for humanitarian logistics practitioners who prioritize reliability and security in aid delivery beyond raw numerical gains.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s6">
<title>6 Conclusion</title>
<p>This paper introduced a hybrid framework integrating Proximal Policy Optimization (PPO), Graph Neural Networks (GNNs), and deterministic constraint validation to optimize humanitarian aid delivery in conflict-affected regions. By combining deep reinforcement learning for adaptive decision-making, graph-based spatial modeling for capturing security risks and logistical dependencies, and structured optimization for feasibility assurance, our approach effectively enhances the efficiency, security, and effectiveness of aid distribution. The framework&#x2019;s effectiveness is demonstrated using real-world georeferenced datasets, including actual road networks from OpenStreetMap and conflict data from the Uppsala Conflict Data Program, with demand modeling incorporating stochastic components to realistically capture operational uncertainties. Experimental results demonstrate that PPO-GNN achieves significant improvements in cost efficiency (7.9% reduction), security risk mitigation (15.17% reduction), and operational reliability (83.33% fewer constraint violations), while substantially improving demand fulfillment compared to traditional DRL and heuristic-based methods. The advantages of this approach become more pronounced in large-scale networks and remain robust even under uncertain conditions, including fluctuating demand, variable security risks, and disruptions in accessibility.</p>
<p>Beyond its quantitative improvements, PPO-GNN offers several practical benefits for humanitarian logistics. Its scalability makes it applicable to both local and national-level operations, while its real-time adaptability enables responsive decision-making in volatile environments. By explicitly modeling security risks and integrating practitioner-informed priorities, the framework aligns with the operational realities faced by humanitarian organizations. This balance between computational efficiency and practical applicability ensures that the proposed approach is not only theoretically sound but also capable of addressing real-world challenges in aid delivery.</p>
<p>However, despite its promising performance, the framework has certain limitations that must be acknowledged. While the framework incorporates real-world road networks and conflict data, the demand modeling includes stochastic components to capture operational uncertainties, meaning that performance in actual conflict zones may vary depending on specific local conditions and data quality. Additionally, the complexity of implementing a hybrid AI-driven approach may pose challenges in resource-constrained humanitarian contexts, requiring potential simplifications for field deployment. Computational requirements, though more efficient than exact solvers, still exceed those of simple heuristics, which could be a limiting factor in environments with restricted computational resources. Moreover, the framework&#x2019;s reliance on hyperparameter tuning may necessitate further research to enhance its adaptability to diverse operational settings. Finally, as with many deep learning-based methods, the interpretability of the model remains a challenge, potentially affecting trust and adoption by practitioners.</p>
<p>Addressing these limitations requires a comprehensive research agenda across multiple dimensions. Real-world validation represents the most pressing priority, requiring close collaboration with humanitarian organizations to conduct field testing in active conflict zones. This approach would provide essential insights into the practical feasibility of the proposed system while identifying necessary adaptations to operational constraints and field conditions.</p>
<p>Model optimization presents another crucial development pathway. Implementing model distillation techniques could significantly simplify policy representations while maintaining performance integrity, thereby facilitating broader adoption across diverse humanitarian contexts. Concurrently, extending the framework to support multi-period planning capabilities that accommodate evolving demand patterns and shifting security conditions would substantially enhance its applicability to the dynamic nature of humanitarian operations.</p>
<p>Computational efficiency improvements could be achieved through transfer learning methodologies, enabling the adaptation of pre-trained models to new geographical regions while reducing deployment costs and accelerating implementation timelines. Furthermore, integrating collaborative decision-making frameworks that incorporate multiple stakeholder perspectives would improve coordination mechanisms and optimize resource allocation in large-scale humanitarian responses. The incorporation of environmental sustainability considerations into logistics planning would align the framework with the growing emphasis on sustainable humanitarian practices.</p>
<p>From a methodological standpoint, future comparative evaluations should encompass advanced heuristic and metaheuristic approaches, including Tabu Search and evolutionary algorithms, alongside emerging hybrid DRL &#x2b; GNN methodologies. While these alternatives show considerable promise, their integration necessitates substantial adaptation and specialized benchmarking efforts tailored to humanitarian vehicle routing challenges. Given these complexities, such extensions are reserved for future investigations to ensure comprehensive and equitable comparative analysis.</p>
<p>These research directions collectively aim to transform the proposed framework from a promising academic contribution into a deployable solution that can significantly impact humanitarian operations worldwide. The PPO-GNN framework represents a meaningful advancement in computational approaches to humanitarian logistics, demonstrating how sophisticated AI methodologies can be adapted to address critical societal challenges. Through continued development and real-world implementation, this work has the potential to enhance the lives of vulnerable populations in conflict-affected regions while advancing the field of humanitarian operations research.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>KM: Conceptualization, Formal Analysis, Methodology, Writing &#x2013; original draft, Writing &#x2013; review and editing. MA: Formal Analysis, Supervision, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. The authors gratefully acknowledge financial support from the Deanship of Scientific Research, King Faisal University (KFU) in Saudi Arabia under grant number KFU253325.</p>
</sec>
<ack>
<p>The authors wish to sincerely thank the Deanship of Scientific Research at King Faisal University for their valuable support of this research. They also express their gratitude to their respective institutions, King Faisal University and the University of Sousse, for providing a conducive academic environment and essential resources. Special thanks are extended to the teams involved in data collection and simulation testing, whose contributions were vital to the successful completion of this study.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. The authors acknowledge the use of Claude 3.7 Sonnet (Anthropic, 2025) to assist with editing portions of this manuscript. All content has been reviewed for factual accuracy by the authors.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Altay</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Green</surname>
<given-names>W. G.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Or/ms research in disaster operations management</article-title>. <source>Eur. J. Operational Res.</source> <volume>175</volume>, <fpage>475</fpage>&#x2013;<lpage>493</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejor.2005.05.016</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balcik</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Beamon</surname>
<given-names>B. M.</given-names>
</name>
<name>
<surname>Smilowitz</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Last mile distribution in humanitarian relief</article-title>. <source>J. Intelligent Transp. Syst.</source> <volume>12</volume>, <fpage>51</fpage>&#x2013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1080/15472450802023329</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bello</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Pham</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>Q. V.</given-names>
</name>
<name>
<surname>Norouzi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Neural combinatorial optimization with reinforcement learning</article-title>. <source>
<italic>Corr.</italic> abs/1611</source>, <fpage>09940</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1611.09940</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Besiou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pedraza-Martinez</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Wassenhove</surname>
<given-names>L. N. V.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Humanitarian operations and the un sustainable development goals</article-title>. <source>Prod. Operations Manag.</source> <volume>30</volume>, <fpage>4343</fpage>&#x2013;<lpage>4355</lpage>. <pub-id pub-id-type="doi">10.1111/poms.13579</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bogyrbayeva</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Meraliyev</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Mustakhov</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Dauletbayev</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Learning to solve vehicle routing problems: a survey</article-title>. <pub-id pub-id-type="doi">10.48550/arXiv.2205.02453</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bozorgi-Amiri</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jabalameli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Al-e-Hashem</surname>
<given-names>S. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>A multi-objective robust stochastic programming model for disaster relief logistics under uncertainty</article-title>. <source>OR Spectr.</source> <volume>35</volume>, <fpage>905</fpage>&#x2013;<lpage>933</lpage>. <pub-id pub-id-type="doi">10.1007/s00291-011-0268-x</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clarke</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wright</surname>
<given-names>J. W.</given-names>
</name>
</person-group> (<year>1964</year>). <article-title>Scheduling of vehicles from a central depot to a number of delivery points</article-title>. <source>Operations Res.</source> <volume>12</volume>, <fpage>568</fpage>&#x2013;<lpage>581</lpage>. <pub-id pub-id-type="doi">10.1287/opre.12.4.568</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dror</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Laporte</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Trudeau</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>Vehicle routing with stochastic demands: properties and solution frameworks</article-title>. <source>Transp. Sci.</source> <volume>23</volume>, <fpage>166</fpage>&#x2013;<lpage>176</lpage>. <pub-id pub-id-type="doi">10.1287/trsc.23.3.166</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fuli</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Foropon</surname>
<given-names>C. R. H.</given-names>
</name>
<name>
<surname>Xin</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Reducing carbon emissions in humanitarian supply chain: the role of decision making and coordination</article-title>. <source>Ann. Operations Res.</source> <volume>319</volume>, <fpage>355</fpage>&#x2013;<lpage>377</lpage>. <pub-id pub-id-type="doi">10.1007/s10479-020-03671-z</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Galindo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Batta</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Review of recent developments in Or/ms research in disaster operations management</article-title>. <source>Eur. J. Operational Res.</source> <volume>230</volume>, <fpage>201</fpage>&#x2013;<lpage>211</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejor.2013.01.039</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holgu&#xed;n-Veras</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>P&#xe9;rez</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Jaller</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wassenhove</surname>
<given-names>L. N. V.</given-names>
</name>
<name>
<surname>Aros-Vera</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>On the appropriate objective function for post-disaster humanitarian logistics models</article-title>. <source>J. Operations Manag.</source> <volume>31</volume>, <fpage>262</fpage>&#x2013;<lpage>280</lpage>. <pub-id pub-id-type="doi">10.1016/j.jom.2013.06.002</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>Z. S.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>A multi-stage stochastic programming model for relief distribution considering the state of road network</article-title>. <source>Transp. Res. Part B Methodol.</source> <volume>123</volume>, <fpage>64</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/j.trb.2019.03.014</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Efficient deep reinforcement learning strategies for active flow control based on physics-informed neural networks</article-title>. <source>Phys. Fluids</source> <volume>36</volume>, <fpage>074112</fpage>. <pub-id pub-id-type="doi">10.1063/5.0213256</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Smilowitz</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Balcik</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Models for relief routing: equity, efficiency and efficacy</article-title>. <source>Transp. Res. Part E Logist. Transp. Rev.</source> <volume>48</volume>, <fpage>2</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1016/j.tre.2011.05.004</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kool</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hoof</surname>
<given-names>H. V.</given-names>
</name>
<name>
<surname>Welling</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Attention, learn to solve routing problems</article-title>,&#x201d; in <source>Proceedings of the international conference on learning representations (ICLR)</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1803.08475</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kov&#xe1;cs</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Spens</surname>
<given-names>K. M.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Humanitarian logistics in disaster relief operations</article-title>. <source>Int. J. Phys. Distribution and Logist. Manag.</source> <volume>37</volume>, <fpage>99</fpage>&#x2013;<lpage>114</lpage>. <pub-id pub-id-type="doi">10.1108/09600030710734820</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Krongauz</surname>
<given-names>D. L.</given-names>
</name>
<name>
<surname>Lazebnik</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Collective evolution learning model for vision-based collective motion with collision avoidance</article-title>. <source>PLoS ONE</source> <volume>18</volume>, <fpage>e0270318</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0270318</pub-id>
<pub-id pub-id-type="pmid">37163523</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kunz</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Wassenhove</surname>
<given-names>L. N. V.</given-names>
</name>
<name>
<surname>Besiou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hambye</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kov&#xe1;cs</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Relevance of humanitarian logistics research: best practices and way forward</article-title>. <source>Int. J. Operations and Prod. Manag.</source> <volume>37</volume>, <fpage>1585</fpage>&#x2013;<lpage>1599</lpage>. <pub-id pub-id-type="doi">10.1108/IJOPM-04-2016-0202</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laporte</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>The vehicle routing problem: an overview of exact and approximate algorithms</article-title>. <source>Eur. J. Operational Res.</source> <volume>59</volume>, <fpage>345</fpage>&#x2013;<lpage>358</lpage>. <pub-id pub-id-type="doi">10.1016/0377-2217(92)90192-C</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lazebnik</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Data-driven hospitals staff and resources allocation using agent-based simulation and deep reinforcement learning</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>126</volume>, <fpage>106783</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2023.106783</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A scenario-based hybrid robust and stochastic approach for joint planning of relief logistics and casualty distribution considering secondary disasters</article-title>. <source>Transp. Res. Part E Logist. Transp. Rev.</source> <volume>141</volume>, <fpage>102029</fpage>. <pub-id pub-id-type="doi">10.1016/j.tre.2020.102029</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Merkulov</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Iceland</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Michaeli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gal</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Barel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shima</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2025</year>). &#x201c;<article-title>Reinforcement-learning-based cooperative dynamic weapon-target assignment in a multiagent engagement</article-title>,&#x201d; in <source>AIAA science and technology forum and exposition, AIAA SciTech forum 2025</source> (<publisher-name>American Institute of Aeronautics and Astronautics AIAA</publisher-name>). <pub-id pub-id-type="doi">10.2514/6.2025-1546</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Najafi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Eshghi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Dullaert</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>A multi-objective robust optimization model for logistics planning in the earthquake response phase</article-title>. <source>Transp. Res. Part E Logist. Transp. Rev.</source> <volume>49</volume>, <fpage>217</fpage>&#x2013;<lpage>249</lpage>. <pub-id pub-id-type="doi">10.1016/j.tre.2012.09.001</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Negi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Humanitarian logistics challenges in disaster relief operations: a humanitarian organisations&#x2019; perspective</article-title>. <source>J. Transp. Supply Chain Manag.</source> <volume>16</volume>. <pub-id pub-id-type="doi">10.4102/jtscm.v16i0.691</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>&#xd6;zdamar</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ertem</surname>
<given-names>M. A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Models, solutions and enabling technologies in humanitarian logistics</article-title>. <source>Eur. J. Operational Res.</source> <volume>244</volume>, <fpage>55</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejor.2014.11.030</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedraza-Martinez</surname>
<given-names>A. J.</given-names>
</name>
<name>
<surname>Wassenhove</surname>
<given-names>L. N. V.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Transportation and vehicle fleet management in humanitarian logistics: challenges for future research</article-title>. <source>EURO J. Transp. Logist.</source> <volume>2</volume>, <fpage>18</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1007/s13676-012-0001-1</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>P&#xe9;rez-Rodr&#xed;guez</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Holgu&#xed;n-Veras</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Inventory-allocation distribution models for post-disaster humanitarian logistics with explicit consideration of deprivation costs</article-title>. <source>Transp. Sci.</source> <volume>50</volume>, <fpage>1261</fpage>&#x2013;<lpage>1285</lpage>. <pub-id pub-id-type="doi">10.1287/trsc.2014.0565</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rodr&#xed;guez-Esp&#xed;ndola</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Albores</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Brewster</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Decision-making and operations in disasters: challenges and opportunities</article-title>. <source>Int. J. Operations and Prod. Manag.</source> <volume>38</volume>, <fpage>1964</fpage>&#x2013;<lpage>1986</lpage>. <pub-id pub-id-type="doi">10.1108/IJOPM-03-2017-0151</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rodr&#xed;guez-Esp&#xed;ndola</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Ahmadi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gast&#xe9;lum-Chavira</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ahumada-Valenzuela</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Chowdhury</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dey</surname>
<given-names>P. K.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Humanitarian logistics optimization models: an investigation of decision-maker involvement and directions to promote implementation</article-title>. <source>Socioecon. Plan. Sci.</source> <volume>89</volume>, <fpage>101669</fpage>. <pub-id pub-id-type="doi">10.1016/j.seps.2023.101669</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wolski</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klimov</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Proximal policy optimization algorithms</article-title>. <pub-id pub-id-type="doi">10.48550/arXiv.1707.06347</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tomasini</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Wassenhove</surname>
<given-names>L. N. V.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>From preparedness to partnerships: case study research on humanitarian logistics</article-title>. <source>Int. Trans. Operational Res.</source> <volume>16</volume>, <fpage>549</fpage>&#x2013;<lpage>559</lpage>. <pub-id pub-id-type="doi">10.1111/j.1475-3995.2009.00697.x</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vega</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Roussat</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Humanitarian logistics: the role of logistics service providers</article-title>. <source>Int. J. Phys. Distribution and Logist. Manag.</source> <volume>45</volume>, <fpage>352</fpage>&#x2013;<lpage>375</lpage>. <pub-id pub-id-type="doi">10.1108/IJPDLM-12-2014-0309</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wassenhove</surname>
<given-names>L. N. V.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Humanitarian aid logistics: supply chain management in high gear</article-title>. <source>J. Operational Res. Soc.</source> <volume>57</volume>, <fpage>475</fpage>&#x2013;<lpage>489</lpage>. <pub-id pub-id-type="doi">10.1057/palgrave.jors.2602125</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wex</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Schryen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Feuerriegel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Neumann</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Emergency response in natural disaster management: allocation and scheduling of rescue units</article-title>. <source>Eur. J. Operational Res.</source> <volume>235</volume>, <fpage>697</fpage>&#x2013;<lpage>708</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejor.2013.10.029</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Kenway</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Mader</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>Jasa</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Martins</surname>
<given-names>J. R. R. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Pyoptsparse: a python framework for large-scale constrained nonlinear optimization of sparse systems</article-title>. <source>J. Open Source Softw.</source> <volume>5</volume>, <fpage>2564</fpage>. <pub-id pub-id-type="doi">10.21105/joss.02564</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yue</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A deep reinforcement learning-based adaptive search for solving time-dependent green vehicle routing problem</article-title>. <source>IEEE Access</source> <volume>12</volume>, <fpage>33400</fpage>&#x2013;<lpage>33419</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2024.3369474</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Physics informed deep reinforcement learning for aircraft conflict resolution</article-title>. <source>IEEE Trans. Intelligent Transp. Syst.</source> <volume>23</volume>, <fpage>8288</fpage>&#x2013;<lpage>8301</lpage>. <pub-id pub-id-type="doi">10.1109/TITS.2021.3077572</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>