<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Signal Process.</journal-id>
<journal-title>Frontiers in Signal Processing</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Signal Process.</abbrev-journal-title>
<issn pub-type="epub">2673-8198</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1608347</article-id>
<article-id pub-id-type="doi">10.3389/frsip.2025.1608347</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Signal Processing</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Reinforcement learning, rule-based, or generative AI: a comparison of model-free Wi-Fi slicing approaches</article-title>
<alt-title alt-title-type="left-running-head">Rosales and Cavalcanti</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frsip.2025.1608347">10.3389/frsip.2025.1608347</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Rosales</surname>
<given-names>Rafael</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2287343/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Cavalcanti</surname>
<given-names>Dave</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3072570/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Intel Labs</institution>, <institution>Intel Corporation</institution>, <addr-line>Munich</addr-line>, <country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Intel CCG</institution>, <institution>Intel Corporation</institution>, <addr-line>Hillsboro</addr-line>, <addr-line>OR</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1771975/overview">Dionysis Kalogerias</ext-link>, Yale University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2808152/overview">Ahmad Bazzi</ext-link>, New York University Abu Dhabi, United Arab Emirates</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3031932/overview">Miguel Calvo-Fullana</ext-link>, Pompeu Fabra University, Spain</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3034541/overview">Sourajit Das</ext-link>, University of Pennsylvania, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Rafael Rosales, <email>rafael.rosales@intel.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>05</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>5</volume>
<elocation-id>1608347</elocation-id>
<history>
<date date-type="received">
<day>08</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>05</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Rosales and Cavalcanti.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Rosales and Cavalcanti</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Resource allocation techniques are key to providing Quality-of-Service guarantees. Wi-Fi standards define features enabling the allocation of radio resources across time, frequency, and link band. However, radio resource slicing, as implemented in 5G cellular networks, is not native to Wi-Fi. A few reinforcement learning (RL) approaches have been proposed for Wi-Fi resource allocation and demonstrated using analytical models where the reward gradient with respect to the model parameters is accessible&#x2014;i.e., with a differentiable Wi-Fi network model. In this work, we implement&#x2014;and release under an Apache 2.0 license&#x2014;a state-of-the-art, state-augmented constrained optimization method using a policy-gradient RL algorithm that does not require a differentiable model, to assess model-free RL-based slicing for Wi-Fi frequency resource allocation. We compare this with six model-free baselines: three RL algorithms (REINFORCE, A2C, PPO), two rule-based heuristics (Uniform, Proportional), and a generative AI policy using a commercial foundational Large Language Model (LLM). For rapid RL training, a simple, non-differentiable network model was used. To evaluate the policies, we use an ns-3-based Wi-Fi 6 simulator with a slice-aware MAC. Evaluations were conducted in two traffic scenarios: A) a periodic pattern with one constant low-throughput slice and two high-throughput slices toggled sequentially, and B) a random walk scenario for realism. Results show that, on average&#x2014;in terms of the trade-off between total throughput and a packet-latency-based metric&#x2014;the uniform split and LLM-based policy perform best, appearing on the Pareto front in both scenarios. The proportional policy only appears on the front in the periodic case. Our state-augmented constrained approach based on REINFORCE (SAC-RE) is on the second Pareto front for the random walk case, outperforming vanilla REINFORCE. In the periodic scenario, vanilla REINFORCE achieves better throughput&#x2014;with a latency trade-off&#x2014;and is co-located with SAC-RE on the second front. Interestingly, the LLM-based policy&#x2014;neither trained nor fine-tuned on any custom data&#x2014;consistently appears on the first Pareto front, offering higher objective values at some latency cost. Unlike uniform slicing, its behavior is dynamically adjustable via prompt engineering.</p>
</abstract>
<kwd-group>
<kwd>optimization</kwd>
<kwd>Wi-Fi</kwd>
<kwd>network slicing</kwd>
<kwd>reinforcement learning</kwd>
<kwd>generative AI</kwd>
<kwd>large language model</kwd>
<kwd>LLM</kwd>
<kwd>state-augmented</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Signal Processing for Communications</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Network slicing is a technique used to guarantee Quality-of-Service (QoS) by allocating resources in a way that allows the network to meet traffic flow requirements. Slicing can be implemented in wireless networks through various mechanisms that leverage the features available in a given technology domain. The CSMA channel access mechanism of Wi-Fi was designed as a solution that provides fairness to all devices, albeit with no guarantees. Slicing allows for the design of a network that trades off this fairness but can provide guarantees in Quality of Service if required. A challenge that arises is that if the slicing policy is not well optimized, it can lead to inefficiencies and waste of resources.</p>
<p>In cellular networks, such as 5G, the 3GPP standards define the capability to schedule channel communication resources in both time and frequency. This capability has been exploited to implement slicing of the radio channel (<xref ref-type="bibr" rid="B24">Zhang, 2019</xref>).</p>
<p>Wi-Fi, on the other hand, has employed a contention-based approach for channel access since its inception, making it more challenging to guarantee QoS. Over time, Wi-Fi standards, defined by the IEEE 802.11 working group, have evolved to offer greater control over channel access, enabling the deployment of centrally managed networks via an Access Point (AP), where deterministic policies can be implemented. However, limited work exists on the implementation of slicing concepts in Wi-Fi. Since Wi-Fi 6, frequency resources can be allocated by the AP to different users. In contrast to 5G cellular standards&#x2014;where Physical Resource Blocks (PRBs) are allocated in both time and frequency domains&#x2014;the AP can only allocate the resources of within a Physical Layer Protocol Unit (PPDU) once it starts a transmission opportunity (TXOP) after gaining channel access; see <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Slicing of Wi-Fi OFDMA frequency resources. Since Wi-Fi 6, an AP can centrally schedule multi-user downlink and uplink transmissions. After acquiring a transmission opportunity (TxOp), the AP can individually assign resource units (RUs) to different client stations (STAs) within a multi-user Physical Protocol Data Unit (PPDU). In this work, we dynamically allocate multiple RUs to create slices in the frequency domain.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g001.tif"/>
</fig>
<p>
<xref ref-type="bibr" rid="B23">Zangooei et al. (2023)</xref>, <xref ref-type="bibr" rid="B21">Yang et al. (2024)</xref>, and <xref ref-type="bibr" rid="B9">Liu et al. (2020)</xref> are examples of Artificial Intelligence (AI) approaches based on Reinforcement Learning (RL) for optimizing resource allocation.</p>
<p>Work incorporating QoS constraints&#x2014;such as <xref ref-type="bibr" rid="B21">Yang et al. (2024)</xref> and <xref ref-type="bibr" rid="B9">Liu et al. (2020)</xref> &#x2014; typically focuses on creating a single objective function by combining multiple constraints and the main objective into a weighted sum. <xref ref-type="bibr" rid="B8">Liu et al. (2021)</xref> explore constraint-aware optimization through Lagrangian primal-dual methods. <xref ref-type="bibr" rid="B3">Calvo-Fullana et al. (2023)</xref> proposed an improvement to these methods using the concept of state-augmented RL, where the values of the Lagrangian multipliers are fed as input to the policy. This enables the learning of different behaviors depending on the current system state.</p>
<p>The state-augmented approach has been adopted by <xref ref-type="bibr" rid="B12">NaderiAlizadeh et al. (2022)</xref> and <xref ref-type="bibr" rid="B19">Uslu et al. (2024)</xref> for radio resource allocation of power transmission levels and frequency resources, respectively. Their work applied state-augmented RL using gradient-based direct optimization, which requires computing the gradient of the Lagrangian w.r.t. the policy parameters&#x2014;i.e., a differentiable model of the system. However, this limits applicability in real systems or simulation environments such as ns-3 (<xref ref-type="bibr" rid="B6">Henderson et al., 2008</xref>), where analytical gradients are not available.</p>
<p>Model-free RL approaches, such as policy-gradient <xref ref-type="bibr" rid="B18">Sutton and Barto (1998)</xref> methods, estimate the gradients via sampling and therefore do not require a differentiable model of the system.</p>
<p>In this work, we evaluate multiple model-free approaches (see <xref ref-type="fig" rid="F2">Figure 2</xref>), including a proposed adaptation of the state-augmented method from <xref ref-type="bibr" rid="B3">Calvo-Fullana et al. (2023)</xref> to the problem of radio frequency resource slicing, as in <xref ref-type="bibr" rid="B19">Uslu et al. (2024)</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>General methodology: RL-based policies are first trained using a simple network simulator. These policies, together with other rule-based and GenAI based policies are evaluated with a ns-3 Wi-Fi simulator.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g002.tif"/>
</fig>
<p>Alongside other RL and rule-based policies, we also evaluate a Generative AI (GenAI) approach using a Large Language Model (LLM) for RL. In this case, a commercial off-the-shelf (COTS) foundational LLM is used to make slicing decisions. LLMs have demonstrated wide applicability due to their emergent properties developed through training on vast amounts of data. <xref ref-type="bibr" rid="B13">Peng et al. (2023)</xref> propose using LLMs as RL policies, offering a natural language interface to specify decisions. <xref ref-type="bibr" rid="B25">Zhou and Small (2021)</xref> apply LLMs to learn policies tailored to specific tasks.</p>
</sec>
<sec id="s2">
<title>2 Network slicing in Wi-Fi</title>
<sec id="s2-1">
<title>2.1 Wi-Fi radio resource allocation features</title>
<p>Wi-Fi provides some alternatives for slicing the communication channel between different users.</p>
<p>The simplest approach is to separate different flows across different frequency bands or channels&#x2014;for example, assigning three different flows to the 2.4 GHz, 5&#xa0;GHz, and 6&#xa0;GHz bands, respectively. However, this method is typically statically configured <italic>a priori</italic> and can be inefficient and prone to Quality-of-Service (QoS) degradation due to dynamic channel fluctuations. Furthermore, it is only feasible in multi-radio devices supporting multiple frequency bands.</p>
<p>At a more granular level, since Wi-Fi 6 it is possible to dynamically assign different bandwidths to different flows by splitting a single channel into multiple Resource Units (RUs) in a multi-user PPDU, thereby distributing the available spectrum more efficiently. For example, a 20&#xa0;MHz channel can be divided into nine RUs, each 2&#xa0;MHz wide. The decision regarding the number and size of RUs in both downlink and uplink transmissions is dynamically made by the Wi-Fi Access Point (AP). Since Wi-Fi 7, this includes support for aggregating multiple RUs for a single user.</p>
<p>In addition to frequency-domain slicing (<xref ref-type="bibr" rid="B19">Uslu et al., 2024</xref>), traffic flows can also be isolated in the time domain. <xref ref-type="bibr" rid="B4">Candell et al. (2022)</xref> demonstrated an implementation of the TSN standard for time-aware scheduling (802.1Qbv) in Wi-Fi. Time-domain allocation must also be centrally managed by the AP. It is important to note that, unlike cellular technologies defined by 3GPP, Wi-Fi does not incorporate the concept of Physical Resource Blocks (PRBs), which enable simultaneous allocation in both time and frequency domains. 3GPP cellular networks have always used a centralized scheduling mechanism, allowing for precise allocation of time slots and frequency bands to different users as part of the protocol. Wi-Fi networks, on the other hand, have been based on a decentralized, contention-based access method. Recently, in Wi-Fi 6, the OFDMA scheduling mechanism was introduced, allowing for the assignment of frequency resources within a PPDU. This solution is not as flexible as the 3GPP approach. In Wi-Fi, time-based resource allocation must be done across multiple PPDUs, thus combining it with the decentralized CSMA mechanism for resource contention. To eliminate the contention overhead, a centrally managed Wi-Fi network is assumed for the proposed Wi-Fi radio slicing; that is, all client devices are centrally managed by the AP using OFDMA. Under this assumption, most of the contention overhead due to channel access is eliminated.</p>
<p>In this work, we explore dynamic slicing of Wi-Fi frequency resources&#x2014;i.e., at each transmission opportunity, the AP decides how many frequency resources to allocate to each of the required slices.</p>
<p>The modulation constellation is determined according to the Wi-Fi standard and is dynamically adjusted based on the available bandwidth of the allocated RUs. The network is configured to use a channel bandwidth of 80&#xa0;MHz, and the transmission power is set at 20 dBm. The propagation model used is the IEEE 802.11 D channel model. Other potential physical layer features that were not explored in our study include: a) spatial streams using MIMO, as we used a single antenna for both transmission and reception; and b) preamble puncturing, which was not explored.</p>
</sec>
<sec id="s2-2">
<title>2.2 Problem formulation</title>
<p>We formulate the slicing problem for Wi-Fi OFDMA frequency resources as a constrained optimization problem:</p>
<p>The goal is to find a slicing decision policy <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> that takes an action <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> based on the system state <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> that maximizes the expected value of a main objective function <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which is here defined as the total amount of bytes received, and which is a function of the system state <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the slicing decision <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> at each time step <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. At the same time, the slicing policy should try to fulfill a set of <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> constraints defined by a list of inequalities <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are the constraint functions, and <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the constraint specifications. The formal formulation is shown in <xref ref-type="disp-formula" rid="e1">Equation 1</xref>:<disp-formula id="e1">
<mml:math id="m13">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mspace width="0.22em"/>
<mml:mspace width="0.22em"/>
<mml:mspace width="0.22em"/>
<mml:munder>
<mml:mrow>
<mml:mi>lim</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mtext>s.t.</mml:mtext>
<mml:mspace width="0.22em"/>
<mml:mspace width="0.22em"/>
<mml:mspace width="0.22em"/>
<mml:munder>
<mml:mrow>
<mml:mi>lim</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>J</mml:mi>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>In this work, we define the system state <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as a vector <inline-formula id="inf14">
<mml:math id="m15">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. This vector represents the traffic demand in terms of number of packets ready to be transmitted at each slice (with a total of <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> slices) for the last last <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> steps.</p>
<p>The policy action <inline-formula id="inf17">
<mml:math id="m18">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> correspond to the slicing decisions, and is defined as a vector <inline-formula id="inf18">
<mml:math id="m19">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, such that <inline-formula id="inf19">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., it defines what proportion of resources, in this case Wi-Fi RUs, are assigned to each slice <inline-formula id="inf20">
<mml:math id="m21">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at each time step <inline-formula id="inf21">
<mml:math id="m22">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. In order to enable the finest granularity possible, we select the smallest possible bandwidth for all RUs, and then aggregate them according to the slicing decision. In contrast to a real Wi-Fi implementation, where the bandwidths of the aggregated RUs would be combined for better efficiency, our simulation-based evaluation bundles multiple individual RUs of the same smallest bandwidth. The proposed allocation of multiple RUs for different users is already supported in the Wi-Fi 7 standard (802.11be).</p>
<p>The underlying system dynamics are treated here as a black box, as the explicit modeling of state transitions in the Wi-Fi network is infeasible. This means that while we do not specify an explicit transition model in our objective function, we implicitly rely on the system&#x2019;s ability to learn these transitions through interaction with the environment. In addition, the problem formulation makes no assumption about the stationarity of the transitions. Thus, only solutions applying dynamic adaptation mechanisms would be effective in a non-stationary environment.</p>
<p>As a constraint, we define a metric based on the average latency of the received packets. The metric adds the individual latencies of each packet as measured from the time of transmission until the time of reception at the MAC layers. For each packet that was not received at all, e.g., because no resource was allocated for the corresponding slice, we include a penalty factor of 100. This penalty is included in this metric to avoid encouraging seemingly low average latency solutions that in reality are filling the transmission queues.</p>
</sec>
<sec id="s2-3">
<title>2.3 Solution approaches</title>
<p>Several optimization approaches can be applied to solve <xref ref-type="disp-formula" rid="e1">Equation 1</xref>. The works of <xref ref-type="bibr" rid="B19">Uslu et al. (2024)</xref> and <xref ref-type="bibr" rid="B12">NaderiAlizadeh et al. (2022)</xref> represent two state-of-the-art RL-based methods that address a closely related problem to slicing wireless communication resources. Both approaches apply the concept of state-augmentation proposed by <xref ref-type="bibr" rid="B3">Calvo-Fullana et al. (2023)</xref> to enable multi-modal solutions, where the system may have more than one optimal operating mode. However, these methods have been applied using a differentiable model of the system in order to guide the learning of the policy as dictated by the gradient of the Lagrangian formulation of the optimization problem.</p>
<p>In this work, we explore the implementation of state-augmentation using a policy-gradient, model-free RL approach, i.e., the system does not need to be a differentiable system model. Policy-gradient methods estimate the gradient of the objective function w.r.t. the policy parameters through sampling. While this approach is less sample-efficient and typically results in slower learning, it has the advantage of being applicable directly to real world systems or high-fidelity simulators such as ns-3. On the other hand, verifying the convergence of policy gradient methods in complex environments such as Wi-Fi slicing poses significant challenges due to non-stationarity, high dimensionality, delayed rewards, and the risk of local optima. These environments require algorithms capable of handling dynamic conditions and vast state-action spaces while ensuring consistent policy improvement. Addressing these theoretical challenges is crucial for effectively deploying reinforcement learning solutions in real-world applications like Wi-Fi slicing.</p>
<p>We also evaluate other common policy-gradient RL methods, including the REINFORCE algorithm, along with simple rule-based heuristics and a policy based on a recent large language model. <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates these approaches in a Venn-diagram, comparing the methods evaluated in this work with the closest related state-of-the-art.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Classification of the approaches evaluated and of the closest related work.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g003.tif"/>
</fig>
<p>In the next subsection, we describe each of the evaluated approaches in more detail.</p>
<sec id="s2-3-1">
<title>2.3.1 Policy-gradient based RL approaches</title>
<p>We evaluate three commonly adopted policy-gradient based RL approaches:</p>
<sec id="s2-3-1-1">
<title>2.3.1.1 REINFORCE algorithm</title>
<p>This method, originally proposed in <xref ref-type="bibr" rid="B20">Williams (1992)</xref> and summarized in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>, consists of two main phases: a) a sampling phase, where a full episode is executed using the current policy <inline-formula id="inf22">
<mml:math id="m23">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to collect <inline-formula id="inf23">
<mml:math id="m24">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> samples of the system state <inline-formula id="inf24">
<mml:math id="m25">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the selected action <inline-formula id="inf25">
<mml:math id="m26">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and corresponding system rewards <inline-formula id="inf26">
<mml:math id="m27">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>; and b) an evaluation phase, where a loss is computed based on the negative log-probability of the sampled actions, scaled by the observed rewards. The evaluation phase encourages updates to the policy parameters such that unlikely decisions with high positive rewards become more probable, while decisions that resulted in negative rewards become less likely.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>REINFORCE algorithm - <xref ref-type="bibr" rid="B20">Williams (1992)</xref>.<list list-type="simple">
<list-item>
<p>
<bold>Require:</bold> A differentiable policy parameterization <inline-formula id="inf27">
<mml:math id="m28">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<bold>Require:</bold> Parameters: learning step size <inline-formula id="inf28">
<mml:math id="m29">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, reward discount factor <inline-formula id="inf29">
<mml:math id="m30">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, total episode steps <inline-formula id="inf30">
<mml:math id="m31">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="double-struck">N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>1:&#x2003;Initialize policy parameters: <inline-formula id="inf31">
<mml:math id="m32">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>2:&#x2003;<bold>for</bold> each episode <bold>do</bold>
</p>
</list-item>
<list-item>
<p>3:&#x2003;&#x2003;Generate: <inline-formula id="inf32">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> following <inline-formula id="inf33">
<mml:math id="m34">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>4:&#x2003;&#x2003;<bold>for</bold> each episode step <inline-formula id="inf34">
<mml:math id="m35">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>5:&#x2003;&#x2003;&#x2003;Compute discounted returns: <inline-formula id="inf35">
<mml:math id="m36">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>6:&#x2003;&#x2003;&#x2003;Update policy parameters: <inline-formula id="inf36">
<mml:math id="m37">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi>G</mml:mi>
<mml:mi>&#x2207;</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>7:&#x2003;&#x2003;<bold>end for</bold>
</p>
</list-item>
<list-item>
<p>8:&#x2003;<bold>end for</bold>
</p>
</list-item>
<list-item>
<p>9:&#x2003;<bold>return</bold> <inline-formula id="inf37">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
<sec id="s2-3-1-2">
<title>2.3.1.2 Synchronous advantage Actor-critic (A2C)</title>
<p>One of the main challenges of RE, is selecting an appropriate baseline to determine whether the observed reward should be interpreted as positive or negative. The A2C approach - a synchronous variant of <xref ref-type="bibr" rid="B11">Mnih et al. (2016)</xref> - implements a value-based model (the critic) to estimate this baseline and uses the difference between the actual reward and the predicted baseline, known as the advantage, in the loss function. We use the Python implementation provided by <xref ref-type="bibr" rid="B17">Stable-Baselines3 (2025b)</xref> for this algorithm. The n_step hyperparameter was set to match the update rate of the other RL algorithms. Other hyper-parameters used the default values.</p>
</sec>
<sec id="s2-3-1-3">
<title>2.3.1.3 Proximal policy optimization</title>
<p>This approach (<xref ref-type="bibr" rid="B15">Schulman et al., 2017</xref>), uses a clipped surrogate loss within a trust region to prevent drastic updates to the policy parameters during training. We use the Python implementation provided by <xref ref-type="bibr" rid="B16">Stable-Baselines3 (2025a)</xref> for this algorithm. Again, the n_step hyperparameter was set to match the update rate of the other RL algorithms. Other hyper-parameters used the default values.</p>
</sec>
</sec>
<sec id="s2-3-2">
<title>2.3.2 Policy-gradient based, state-augmented constrained RL</title>
<p>We implemented a state-augmented implementation of the REINFORCE algorithm. In the remainder of this paper, we refer to this implementation as SAC-RE. Below, we describe the approach and model in detail.</p>
<sec id="s2-3-2-1">
<title>2.3.2.1 State-augmented constrained REINFORCE</title>
<p>
<xref ref-type="bibr" rid="B3">Calvo-Fullana et al. (2023)</xref> introduced the concept of state-augmentation for constrained reinforcement learning. The core idea is to make the policy model <inline-formula id="inf38">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> a function not only of the system state <inline-formula id="inf39">
<mml:math id="m40">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, but also of the Lagrangian multipliers: <inline-formula id="inf40">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The Lagrangian multipliers <inline-formula id="inf41">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> where <inline-formula id="inf42">
<mml:math id="m43">
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the index of the associated constraint, are then used to integrate the <inline-formula id="inf43">
<mml:math id="m44">
<mml:mrow>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> constraints of the problem into a single objective function <inline-formula id="inf44">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="equ1">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>where <inline-formula id="inf45">
<mml:math id="m47">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the original reward, <inline-formula id="inf46">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the <italic>j</italic>th constraint cost signal, and <inline-formula id="inf47">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the associated constraint threshold.</p>
<p>By doing so, the policy can learn different behaviors depending on the current situation or mode, thereby enabling the learning of optimal solutions under varying circumstances.</p>
<p>While REINFORCE, A2C, and PPO struggle to adapt swiftly to abrupt traffic changes due to their update strategies that rely on episodic or incremental updates, the state-augmented approach with Lagrangian multipliers can enhance them. This modification allows for dynamic adjustment and improved responsiveness as soon as a constraint is violated, offering better adaptability in fast-changing environments compared to the slower traditional methods. The Lagrangian formulation of the reward objective allows to dynamically adjust how much importance (or weight) is given to each objective based on how well constraints are being met during learning. Instead of using fixed weights for each objective, the RL algorithm modifies these weights as necessary, aiming to find a solution that minimizes constraint violations. This dynamic adjustment introduces some uncertainty because the specific tradeoff between objectives depends on how the neural network was initialized and trained.</p>
<p>The state-augmented approach defines a training algorithm for the policy, which can incorporate different reinforcement learning methods, and an inference algorithm, that updates the Lagrangian multipliers at runtime.</p>
<p>
<statement content-type="algorithm" id="Algorithm_2">
<label>Algorithm 2</label>
<p>State-augmented training algorithm - <xref ref-type="bibr" rid="B3">Calvo-Fullana et al. (2023)</xref>.<list list-type="simple">
<list-item>
<p>1: Sample <inline-formula id="inf48">
<mml:math id="m50">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> from augmented space <inline-formula id="inf49">
<mml:math id="m51">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi mathvariant="normal">&#x39b;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>2: Construct augmented rewards <inline-formula id="inf50">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>J</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>3: Use an RL algorithm to obtain policy <inline-formula id="inf51">
<mml:math id="m53">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>lim</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
</list>
</p>
</statement>
</p>
<p>
<statement content-type="algorithm" id="Algorithm_3">
<label>Algorithm 3</label>
<p>State-augmented inference algorithm - <xref ref-type="bibr" rid="B3">Calvo-Fullana et al. (2023)</xref>.<list list-type="simple">
<list-item>
<p>
<bold>Require:</bold> Policy <inline-formula id="inf52">
<mml:math id="m54">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, step <inline-formula id="inf53">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, requirements <inline-formula id="inf54">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, epoch <inline-formula id="inf55">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>1:&#x2003;Initialize: Given initial state <inline-formula id="inf56">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, dual variables <inline-formula id="inf57">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>2:&#x2003;<bold>for</bold> k &#x3d; 0,1, &#x2026; ,K <bold>do</bold>
</p>
</list-item>
<list-item>
<p>3:&#x2003;&#x2003;Rollout <inline-formula id="inf58">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> steps with actions <inline-formula id="inf59">
<mml:math id="m61">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>J</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>4:&#x2003;&#x2003;Update dual variables <inline-formula id="inf60">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>k</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>5:&#x2003;<bold>end for</bold>
</p>
</list-item>
</list>
</p>
</statement>
</p>
<p>We implemented Step 3 of the State-augmented Training algorithm (<xref ref-type="statement" rid="Algorithm_2">Algorithm 2</xref>) using the REINFORCE algorithm (<xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>). The inference time algorithm (<xref ref-type="statement" rid="Algorithm_3">Algorithm 3</xref>) was implemented as originally proposed. For REINFORCE, we adopted a mean baseline - i.e., we subtracted the mean of the episode&#x2019;s rewards from each observed reward.</p>
<p>The slicing policy model <inline-formula id="inf61">
<mml:math id="m63">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is parameterized by the weights <inline-formula id="inf62">
<mml:math id="m64">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of a multi-layer perceptron (MLP) composed of three linear layers, each followed by a leaky ReLU activation function. The number of inputs corresponds to the number of system state signals multiplied by a factor <inline-formula id="inf63">
<mml:math id="m65">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, as the network receives not only the current system state but also the previous <inline-formula id="inf64">
<mml:math id="m66">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> states. The second linear layer has dimensions <inline-formula id="inf65">
<mml:math id="m67">
<mml:mrow>
<mml:mn>128</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>128</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, and the final linear layer has dimensions <inline-formula id="inf66">
<mml:math id="m68">
<mml:mrow>
<mml:mn>128</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf67">
<mml:math id="m69">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of slices.</p>
<p>The output of the neural network is used as the set of alpha parameters for a Dirichlet distribution. The Dirichlet distribution parameterizes a probability distribution over components that sum to one, making it suitable for representing resource slicing decisions. Depending on the alpha values, the distribution may favor uniform splits, skewed allocations, or evenly distributed probabilities across possible combinations. A key benefit of using the Dirichlet distribution is that it encourages exploration of diverse slicing combinations - since the model predicts a probability distribution rather than a fixed decision. A deterministic approach using a Softmax function would produce more precise, single-point predictions for resource allocation. While this might simplify decision-making and reduce computational complexity, it could also lead to less robust performance under varying network conditions because it does not inherently capture uncertainty or variability in the same way that a Dirichlet-based probabilistic model does. In our experiments, the alpha parameters of the Dirichlet distribution were constrained to values between 1 and 10,000, allowing the distribution to be flat or moderately peaked, but avoiding excessively sharp outputs.</p>
<p>We used the Adam optimizer (<xref ref-type="bibr" rid="B7">Kingma and Ba, 2015</xref>) to update the neural network parameters, with a learning rate of <inline-formula id="inf68">
<mml:math id="m70">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.0001</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>The source code of the SAC-RE implementation is publicly available in <xref ref-type="bibr" rid="B14">Rosales (2025)</xref> under the Apache 2.0 License.</p>
</sec>
</sec>
<sec id="s2-3-3">
<title>2.3.3 Rule-based approaches</title>
<p>Typical baselines for comparing RL approaches rely on static heuristics, which aim to balance multiple objectives and constraints using simple, rule-based strategies that require no training.</p>
<sec id="s2-3-3-1">
<title>2.3.3.1 Uniform heuristic</title>
<p>This heuristic evenly divides the available resources among all slices. For example, if three slices are defined, it always produces a split of <inline-formula id="inf69">
<mml:math id="m71">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0.33</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.33</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.33</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. While this approach provides a basic level of fairness - regardless of scenario complexity - it can be inefficient and lead to resource under utilization in dynamic or asymmetric traffic conditions.<disp-formula id="e2">
<mml:math id="m72">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf70">
<mml:math id="m73">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the allocated amount of resources for slice <inline-formula id="inf71">
<mml:math id="m74">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf72">
<mml:math id="m75">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of slices.</p>
</sec>
<sec id="s2-3-3-2">
<title>2.3.3.2 Proportional heuristic</title>
<p>This heuristics determines the slicing decision based on a signal of interest - such as current traffic demand - and allocates resources to each slice in proportion to its share of that signal.<disp-formula id="e3">
<mml:math id="m76">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf73">
<mml:math id="m77">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the allocated amount of resources for slice <inline-formula id="inf74">
<mml:math id="m78">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf75">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the signal of interest of slice <inline-formula id="inf76">
<mml:math id="m80">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf77">
<mml:math id="m81">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the total number of slices.</p>
</sec>
</sec>
<sec id="s2-3-4">
<title>2.3.4 Generative AI approach</title>
<sec id="s2-3-4-1">
<title>2.3.4.1 LLM heuristic</title>
<p>Large Language Models (LLMs) have recently gained significant attention due to their emergent capabilities. An LLM is a neural network trained at scale in a self-supervised manner on vast amounts of text data, with the goal of predicting the most likely next word in a sequence. These models can contain billions of parameters, and despite their relatively simple training objective, they are capable of generating highly plausible and contextually appropriate text across a wide range of domains.</p>
<p>Recently, LLMs have begun to be integrated as decision-making components within RL frameworks, such as <xref ref-type="bibr" rid="B13">Peng et al. (2023)</xref>. In this work, we evaluate the performance of a foundational LLM as a slicing policy - i.e., an LLM that has not been fine-tuned or optimized for a specific task, but rather used originally pre-trained. An LLM can be instructed to solve a specific task by providing the necessary context in the form of a <italic>prompt</italic> - the text input fed to the model. With a detailed and well-crafted prompt, an LLM can be conditioned to produce outputs in a desired format or domain. An entire area of research is devoted to prompt engineering - designing optimal prompts to elicit the best possible responses from an LLM. However, in this paper, prompt optimization is out of scope. We focus on evaluating a single LLM: GPT-4o, using a single detailed prompt template, denoted as <inline-formula id="inf78">
<mml:math id="m82">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>m</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="boxed-text" rid="dBox1">Box 1</xref>). This template is used each time a new slicing decision is required, i.e., at every policy inference step, with the placeholder <inline-formula id="inf79">
<mml:math id="m83">
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> dynamically replaced by the current system state. This means, that the system state is fed to the LLM prompt in the <inline-formula id="inf80">
<mml:math id="m84">
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> placeholder.</p>
<boxed-text id="dBox1">
<label>BOX 1</label>
<title>Slicing decision prompt: <inline-formula id="inf81">
<mml:math id="m85">
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>m</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</title>
<p>&#x201c;Prompt: You are an intelligent resource allocation agent responsible for distributing frequency resources among three network slices. Your goal is to optimize resource allocation dynamically based on traffic demand.</p>
<p>Input: A three-dimensional vector representing the current traffic demand for the three slices. A history of the last five traffic demands, forming an input state of size 18 (3 &#xd7; 6).</p>
<p>Output: A three-dimensional vector representing the proportion of available bandwidth allocated to each slice. The sum of the three components must always equal 1 (i.e., the full bandwidth is allocated).</p>
<p>Decision Objective: Prioritize slices with higher traffic demand while ensuring fairness and avoiding excessive fluctuations. Adapt dynamically to historical trends in traffic to prevent congestion. Avoid under-utilization or over-allocation of resources to any slice. Protect the time-critical slice &#x23;2 so that packets in this slice can always be transmitted.</p>
<p>Constraints: Sum constraint: The three output values must sum to 1. Non-negativity: Each output component must be <inline-formula id="inf82">
<mml:math id="m86">
<mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Responsiveness: The allocation should react to changing demands rather than relying solely on past data.</p>
<p>Example Input-Output Pair: Input: [0.5, 0.25, 0.25, 0.45, 0.3, 0.25, 0.4, 0.35, 0.25, 0.38, 0.32, 0.3, 0.4, 0.3, 0.3] (Flattened to a single 18-dimensional vector, where the current demand vector is: [0.4, 0.3, 0.3]).</p>
<p>Expected Output: A valid allocation such as [0.42, 0.3, 0.28] (ensuring sum &#x3d; 1).</p>
<p>Provide the optimal allocation as a three-dimensional vector. Provide context as needed, but the final answer must always be placed on a single separate line at the very end inside square brackets (do not use code blocks). Write nothing else after that!</p>
<p>Input: {data}&#x201d;</p>
</boxed-text>
<p>The prompt provided to the LLM is designed to enforce that the final answer appears at the end in a specific format. However, a parser is still required to handle slight variations in the output - such as the decision vector being followed by a period or enclosed within a markdown code block. Once the string corresponding to the slicing decision is extracted, it is cast into a tensor for integration into the simulator.</p>
</sec>
</sec>
</sec>
<sec id="s2-4">
<title>2.4 Simulation-based evaluation</title>
<sec id="s2-4-1">
<title>2.4.1 Simulator-based training</title>
<p>To evaluate the different slicing approaches, we implemented a Wi-Fi 6 slice-aware MAC layer in the ns-3 simulator, serving as a proxy for a real Wi-Fi system. Our implementation allows configuration of the number of RUs allocated in the downlink for each slice The AP transmits a packet only if the corresponding flow belongs to a slice with available RUs for transmission.</p>
<p>To accelerate the training of multiple epochs across different RL algorithms, we also implemented a simplified network simulator in Python, which does not implement any Wi-Fi protocol overhead. This simplified simulator is still treated as a black-box model, meaning that reward gradients w.r.t. the policy parameters are not accessible.</p>
</sec>
<sec id="s2-4-2">
<title>2.4.2 Model selection</title>
<p>During training, snapshots of the model weights are stored to allow selection of the best-performing model for inference once training is complete. Model selection is based on the observed reward and constraint values. Specifically, the selection heuristic identifies the training step at which the reward function is maximized while constraint violations are minimized. Since the reward and constraints may be in a trade-off relationship, the selected snapshot corresponds to the point where the reward ceases to improve, or earlier if the constraint functions begin to noticeably increase.</p>
</sec>
<sec id="s2-4-3">
<title>2.4.3 Synchronization between ns-3 simulator and RL framework</title>
<p>To enable synchronization between the ns-3 simulator and the slicing policy, we employed the ns3-ai framework (<xref ref-type="bibr" rid="B22">Yin et al., 2020</xref>). ns3-ai provides a shared-memory mechanism for transferring information between C&#x2b;&#x2b; and Python processes via a Gymnasium (<xref ref-type="bibr" rid="B2">Brockman et al., 2016</xref>) RL interface. This allows the Wi-Fi model to be wrapped as a Gym environment, enabling seamless integration with Python-based RL workflows.</p>
<p>From the Gym perspective, an RL policy is executed step-by-step within an episode. From the ns-3 perspective, a heartbeat process was implemented to periodically monitor the required signals - such as system state and constraints - and to read the slicing decisions. In the simulation-based evaluation, we assume the ideal case where the latency of each policy is negligible, as simulated time does not advance during policy inference calls.</p>
<p>Before the first step can be performed, an initial setup period is allowed to be simulated in ns-3. During this period, initialization certain processes - such as the ARP protocol to map IP to MAC addresses - are allowed to complete. After initialization, the heartbeat interval is set to 100 milliseconds, meaning that 10 Gym steps correspond to one second of simulated time. For each training epoch in the Gym environment, a complete episode is simulated in ns-3. Each RL approach was trained for a total of 2400 steps.</p>
<p>In SAC-RE, during inference, we used <inline-formula id="inf83">
<mml:math id="m87">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, which indicates that the state augmented variables are updated every four inference steps.</p>
</sec>
</sec>
<sec id="s2-5">
<title>2.5 Test scenarios</title>
<p>This section describes the evaluated topology, application flows, and traffic generation patterns.</p>
<p>The topology consists of a single Access Point (AP) with three associated stationary client stations (STAs). Application flows are defined as downlink streams from the AP to each of the three STAs. Three slices are defined and mapped one-to-one to the STA links, such that all traffic destined for a given STA corresponds to a single slice.</p>
<sec id="s2-5-1">
<title>2.5.1 Traffic generation</title>
<p>During the setup phase, some baseline flows - such as ARP - are established to enable simulation of Wi-Fi OFDMA downlink transmissions.</p>
<p>Two traffic generation scenarios are defined for the application flows:</p>
<sec id="s2-5-1-1">
<title>2.5.1.1 Periodic pattern</title>
<p>A simple, easily visualizable traffic pattern is first defined to facilitate manual verification. One flow maintains a constant low throughput, representing a low-latency slice. The other two flows alternate periodically between generating high-throughput traffic and minimal traffic. A visualization of this pattern is shown in <xref ref-type="fig" rid="F6">Figure 6</xref>.</p>
</sec>
<sec id="s2-5-1-2">
<title>2.5.1.2 Random walk pattern</title>
<p>The second scenario involves three flows, each starting with the same initial throughput and evolving according to a random walk. At each time step, the traffic of each flow increases or decreases by a value sampled from a uniform distribution between &#x2212;500 and 500 packets. The resulting traffic is constrained between a minimum of 0 and a maximum of 4,000 packets per time step. A visualization of this pattern is shown in <xref ref-type="fig" rid="F7">Figure 7</xref>.</p>
</sec>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<p>This section presents the training and model selection procedure for the implemented SAC-RE policy. We also present the inference results for all reinforcement learning-based policies, including SAC-RE, along with the results of the policies that do not require any training.</p>
<sec id="s3-1">
<title>3.1 Training</title>
<p>As described in <xref ref-type="sec" rid="s2-3-2">Section 2.3.2</xref>, <xref ref-type="sec" rid="s2-3-1">Section 2.3.1</xref>, the training of the reinforcement learning approaches was conducted using a simple Python-based simulator, separately for the two traffic scenarios outlined in <xref ref-type="sec" rid="s2-5">Section 2.5</xref>. All models were trained on a machine equipped with an 11th Gen Intel(R) Core(TM) i7-11850H @ 2.50&#xa0;GHz CPU and an NVIDIA RTX A3000 GPU.</p>
<p>The training times for each policy are summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Training times for the different slicing policies under the periodic traffic pattern. Note that the rule-based and LLM-based policies were not trained on any traffic pattern.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Slicing policy</th>
<th align="center">SAC-RE</th>
<th align="center">RE</th>
<th align="center">A2C</th>
<th align="center">PPO</th>
<th align="center">Uniform</th>
<th align="center">Proportional</th>
<th align="center">LLM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Training Time (hrs)</td>
<td align="center">1.18</td>
<td align="center">1.28</td>
<td align="center">0.433</td>
<td align="center">0.449</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">0</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3-1-1">
<title>3.1.1 SAC-RE learning convergence</title>
<p>
<xref ref-type="fig" rid="F4">Figure 4</xref> illustrates the learning stability of the SAC-RE training procedure by showing the statistic results of the objective function across ten independent runs of the periodic traffic scenario, all using the same set of hyper-parameters.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Training convergence of the SAC-RE policy across 10 independent training runs in the periodic traffic scenario. The x-axis represents the training step (averaged over intervals of 1000 steps), and the y-axis shows the total number of received bytes. The observed pronounced spikes result from the injected traffic pattern. The learning process of different instances begins with a higher variance in the reward signal. Over time, the averaged interquartile range tends to shrink, maintaining a consistent appearance at points where the traffic pattern changes.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g004.tif"/>
</fig>
<p>The training convergence plot indicates that SAC-RE consistently learns an effective policy in the periodic scenario, exhibiting a reward progression over time that aligns with the underlying traffic pattern.</p>
</sec>
<sec id="s3-1-2">
<title>3.1.2 Selection of best model</title>
<p>During the training of all reinforcement learning policies, snapshots of the model weights are periodically saved to allow selection of the best model achieved so far. <xref ref-type="fig" rid="F5">Figure 5</xref> illustrates the manual selection procedure described in <xref ref-type="sec" rid="s2-4">Section 2.4</xref>, used to determine which model snapshot to use for inference. As highlighted by the orange dotted line in both plots, the selected training step corresponds to a point where the objective function has plateaued, while the average latency penalty has not yet started to increase. This choice avoids selecting a model that excessively prioritizes the objective function at the cost of violating system constraints.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Model selection - This figure shows the objective function (left) and system constraints (right) of the RE policy during the training procedure. The x-axis represents training steps (downsampled by &#xd7;100 for visualization). The orange dotted line indicates the manually selected model snapshot used for inference.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g005.tif"/>
</fig>
</sec>
</sec>
<sec id="s3-2">
<title>3.2 Slicing evaluations with model-free approaches in ns-3</title>
<p>In this section, the inference results of all evaluated policies are presented. The inference times for each policy are shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Inference time statistics for slicing decision across different policies. Note that response time of the commercial LLM is in the order of seconds, with high variability depending on current service demand.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Slicing policy</th>
<th align="center">SAC-RE</th>
<th align="center">RE</th>
<th align="center">A2C</th>
<th align="center">PPO</th>
<th align="center">Uniform</th>
<th align="center">Proportional</th>
<th align="center">LLM</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Mean inference time (ms) <inline-formula id="inf84">
<mml:math id="m88">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">2.568</td>
<td align="center">4.700</td>
<td align="center">2.568</td>
<td align="center">2.276</td>
<td align="center">0.038</td>
<td align="center">0.092</td>
<td align="center">4,501</td>
</tr>
<tr>
<td align="center">Standard deviation <inline-formula id="inf85">
<mml:math id="m89">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">4.489</td>
<td align="center">0.977</td>
<td align="center">4.488</td>
<td align="center">3.476</td>
<td align="center">0.011</td>
<td align="center">0.109</td>
<td align="center">4,248</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<xref ref-type="fig" rid="F6">Figures 6</xref>, <xref ref-type="fig" rid="F7">7</xref> present the evaluation results of all seven policies under the periodic traffic and random walk traffic scenarios, respectively. In both figures, Column (A) shows the traffic demand per slice, with each slice represented by a distinct color. The x-axis corresponds to the episode step from the RL environment&#x2019;s perspective. As described in <xref ref-type="sec" rid="s2-4-3">Section 2.4.3</xref>, each step represents 0.1&#xa0;s in the NS3 simulation. Two full episodes (each consisting of 100 steps) are displayed. Column (B) shows the slicing decisions made by the policy at each episode step, visualized as a stacked plot. This represents the proportion of radio resources (ranging from 0 to 1) allocated to each slice over time. Column (C) displays the corresponding reward signal, measured as the total amount of received bytes. Column (D) shows the average constraint penalty metric, which, as defined in <xref ref-type="sec" rid="s2-2">Section 2.2</xref>, penalizes all packets that were enqueued but not successfully delivered to their destination.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Comparison of different slicing policies (row-wise) in the periodic traffic scenario. Shown are only the first 200 inference time steps of in the x-axis. Columns: (Traffic demand) Input system state, (Slicing decisions) Policy actions, (Rewards) Total received bytes, and (Average latency penalty) RL Constraint.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Comparison of different slicing policies (row-wise) in the random walk traffic scenario. Only the first 200 inference time steps are shown on the x-axis. Columns: (Traffic demand) Input system state, (Slicing decisions) Policy actions, (Rewards) Total received bytes, and (Average latency penalty) RL Constraint.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g007.tif"/>
</fig>
<sec id="s3-2-1">
<title>3.2.1 Observations to inference results</title>
<p>When comparing the slicing decisions with the system state in <xref ref-type="fig" rid="F6">Figure 6</xref>, it can be observed that SAC-RE tends to allocate most resources to the slice with the highest current demand. For example, note the high percentage of resources allocated to the low-traffic blue slice during the first 20 steps&#x2014;reflecting the absence of competing demand. In contrast, RE does not appear to allocate significant resources to the low-throughput slice, even when no other slice is active. This is evident in the limited allocation to the red slice during the 20&#x2013;40 step interval, despite its being the only active flow.</p>
<p>The other RL approaches, PPO and A2C, failed to effectively adapt to the periodic traffic pattern simulated in ns-3. A2C produced saturated decisions, allocating all resources to a single slice&#x2014;though not consistently to the one with the highest demand. PPO, on the other hand, consistently assigned all resources to the blue slice. These outcomes suggest that both PPO and A2C may require extensive hyperparameter tuning to perform well in this environment.</p>
<p>The LLM-based policy closely mirrors the proportional heuristic, with a notable bias toward slice 2 &#x2014; consistent with the prompt used. However, its relatively high inference time, not accounted for in the simulation, limits its practicality in real-time scenarios. In its current form, an unoptimized commercial off-the-shelf LLM may only be suitable for applications with time constraints on the order of seconds. To make LLMs viable for real-world Wi-Fi slicing applications operating on sub-second timescales, inference latency would need to be significantly reduced using existing optimization techniques.</p>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Policy comparison</title>
<p>To facilitate a clearer comparison of the different policies, <xref ref-type="fig" rid="F8">Figures 8</xref>, <xref ref-type="fig" rid="F9">9</xref> summarize the reward and constraint metrics over time as a single average value per traffic scenario.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Comparison of slicing policies for the periodic traffic pattern. The lower-right region represents the ideal trade-off between high reward and low constraint penalty. Multiple Pareto fronts are plotted using dotted lines, with the first front representing the set of non-dominated solutions and subsequent fronts indicating decreasing levels of optimality.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g008.tif"/>
</fig>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Comparison of slicing policies for the random walk traffic pattern. The lower-right region represents the ideal trade-off between high reward and low constraint penalty. Multiple Pareto fronts are plotted using dotted lines, with the first front representing the set of non-dominated solutions and subsequent fronts indicating decreasing levels of optimality.</p>
</caption>
<graphic xlink:href="frsip-05-1608347-g009.tif"/>
</fig>
<p>The ideal solution is located in the lower-right region of each Figure, representing both high reward and low constraint penalty. Since it is often not possible to simultaneously optimize both objectives in challenging scenarios, a common approach is to visualize the Pareto front to compare the performance of competing solutions. In multi-objective optimization, the Pareto front represents a set of non-dominated solutions&#x2014;that is, no other solution is strictly better in all objectives. A solution is said to dominate another if it improves at least one objective without worsening any other. However, solutions on the Pareto front cannot be ranked relative to each other, as they reflect trade-offs between competing goals. In <xref ref-type="fig" rid="F8">Figures 8</xref>, <xref ref-type="fig" rid="F9">9</xref>, multiple Pareto fronts are plotted sequentially to visualize the performance tiers of each policy.</p>
<sec id="s3-3-1">
<title>3.3.1 Periodic traffic scenario</title>
<p>
<xref ref-type="fig" rid="F8">Figure 8</xref> shows that the uniform, Proportional, and LLM-based policies lie on the first Pareto front, identifying them as dominating solutions. They are followed by SAC-RE and RE on the second Pareto front, and finally by A2C and PPO. In this scenario, the LLM-based policy is positioned at a favorable trade-off point between reward and constraint satisfaction. Note that the y-axis uses a logarithmic scale, which emphasizes that even small horizontal shifts (i.e., in reward) can correspond to substantial changes in the constraint penalty.</p>
</sec>
<sec id="s3-3-2">
<title>3.3.2 Random walk traffic scenario</title>
<p>In the random walk traffic scenario, shown in <xref ref-type="fig" rid="F9">Figure 9</xref>, the uniform and LLM-based policies again dominate, appearing on the first Pareto front. This time, the SAC-RE policy represents the entire second Pareto front, demonstrating stronger relative performance compared to other RL approaches. The RE and Proportional policies fall to the third front, followed by A2C, and finally PPO. Unlike in the period scenario, the Proportional policy no longer qualifies as a Pareto-optimal solution.</p>
</sec>
</sec>
<sec id="s3-4">
<title>3.4 Inspection of a sample LLM response</title>
<p>In this section, we examine the behavior of the LLM-based policy for a single inference instance to better understand the &#x201c;reasoning&#x201d; behind its slicing decision. The sampled system state vector <inline-formula id="inf86">
<mml:math id="m90">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, used as input to the <inline-formula id="inf87">
<mml:math id="m91">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> placeholder in the <italic>llm_policy</italic> prompt template, is given by:<disp-formula id="equ2">
<mml:math id="m92">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mn>0.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>5357.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>535.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>6.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>5357.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>536.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>5.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>5357.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>536.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>6.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>5357.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>535.0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>5.0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>The corresponding full response generated by the LLM for this prompt is shown in <xref ref-type="boxed-text" rid="dBox2">Box 2</xref>:</p>
<boxed-text id="dBox2">
<label>BOX 2</label>
<title>LLM response.</title>
<p>To determine the optimal allocation of resources among the three network slices, let&#x2019;s analyze the given input data, which includes the current traffic demand and the historical traffic demand for the last five intervals.</p>
<p>Input: [0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 5357.0, 535.0, 6.0, 5357.0, 536.0, 5.0, 5357.0, 536.0, 6.0, 5357.0, 535.0, 5.0].</p>
<p>1. The current demand vector is [5357.0, 535.0, 5.0]. 2. The historical demands show a consistent pattern where slice &#x23;1 has the highest demand, followed by slice &#x23;2, and slice &#x23;3 has minimal demand.</p>
<p>Given the decision objective, we need to prioritize slice &#x23;2 to ensure it can always transmit packets, and we should allocate resources based on demand while maintaining fairness and avoiding excessive fluctuations.</p>
<p>Considering the historical trend and current demand:</p>
<p>- Slice &#x23;1 (demand: 5357.0) has the highest demand.</p>
<p>- Slice &#x23;2 (demand: 535.0) must be protected for time-critical packets.</p>
<p>- Slice &#x23;3 (demand: 5.0) has the lowest demand.</p>
<p>We will allocate bandwidth proportionally to the demand, ensuring slice &#x23;2 gets sufficient resources, and then distribute the remaining bandwidth between slices &#x23;1 and &#x23;3.</p>
<p>Let&#x2019;s calculate the allocation:</p>
<p>- Total demand: 5357.0 &#x2b; 535.0 &#x2b; 5.0 &#x3d; 5897.0.</p>
<p>- Allocate a base of 0.3 to slice &#x23;2 (time-critical protection).</p>
<p>- Remaining proportion: 1.0&#x2013;0.3 &#x3d; 0.7.</p>
<p>Allocation based on demand:</p>
<p>- Slice &#x23;1: (5357.0/5897.0) &#x2a; 0.7 <inline-formula id="inf88">
<mml:math id="m93">
<mml:mrow>
<mml:mo>&#x2248;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.636.</p>
<p>- Slice &#x23;2: 0.3 (protected).</p>
<p>- Slice &#x23;3: (5.0/5897.0) &#x2a; 0.7 <inline-formula id="inf89">
<mml:math id="m94">
<mml:mrow>
<mml:mo>&#x2248;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.001.</p>
<p>Final allocation:</p>
<p>- Slice &#x23;1: 0.636.</p>
<p>- Slice &#x23;2: 0.3.</p>
<p>- Slice &#x23;3: 0.064 (adjusting for the remaining sum to 1).</p>
<p>The final allocation vector is: [0.636, 0.3, 0.064].</p>
</boxed-text>
<p>From this response, the parser extracted the following vector as the slicing decision:<disp-formula id="equ3">
<mml:math id="m95">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mn>0.6360</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.3000</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.0640</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>The prompt template used enabled the LLM to &#x201c;reason&#x201d; step by step through the task. This approach helps reduce hallucinations, as prompting for an immediate direct answer may cause the LLM to rely too heavily on memorized patterns from its training data, rather than drawing on higher-level reasoning processes&#x2014;which are likely more robust and broadly represented in its training corpus. As observed in this instance, the heuristic followed by the LLM begins by allocating a fixed base of 30% to slice 2 to ensure a minimum level of resources. The remaining budget is then distributed proportionally across the other slices, based on patterns inferred from the historical system state. However, the LLM proposed an erroneous split of the 70% remainder, as it normalized against the total demand (5897) instead of using only the sum of the demand for slice 1 and slice 3. The model ultimately corrected the flawed formula by allocating more resources to slice 3 to ensure that the sum adds up to 1.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussions</title>
<sec id="s4-1">
<title>4.1 Best slicing policies</title>
<p>Our results in <xref ref-type="fig" rid="F8">Figures 8</xref>, <xref ref-type="fig" rid="F9">9</xref> show that, when averaging across an entire inference scenario&#x2014;in terms of total throughput versus average latency penalty&#x2014;the uniform, proportional, and the LLM-based policies dominate the others in the periodic traffic pattern scenario, with the LLM policy effectively balancing both the reward and latency constraint. In the random walk scenario, the proportional policy is no longer part of the first Pareto front, being surpassed by the SAC-RE RL policy as well. This indicates that the RL based policies were not able to predict the traffic patterns as effectively. Among them, the state-augmented version of REINFORCE SAC-RE performed best. This can be attributed to its ability to model multiple modes of operation and achieve better performance, even without extensive hyper-parameter tuning, thanks to the use of the state augmentation.</p>
<p>The LLM policy on the other hand&#x2014;which was not re-trained nor fine-tuned on any data&#x2014;appears at the first Pareto front in both scenarios, providing a higher value of the objective function at some cost to the latency constraint. Unlike the uniform slicing approach, the LLM policy&#x2019;s behavior can be adjusted via prompt alone. Thus, the LLM policy thus provides a flexible and easy-to-setup slicing approach, as it can leverage the knowledge encoded in the foundational LLM to decide which action to take without requiring an extensive training. The main cost of this approach lies in the expensive inference calls, both in terms of compute resources and inference latency. However, recent progress in knowledge distillation (<xref ref-type="bibr" rid="B1">Acharya et al., 2024</xref>) and model pruning of LLMs into smaller models (<xref ref-type="bibr" rid="B10">Ma et al., 2023</xref>) has made promising steps towards mitigating high latency in LLMs. In terms of inference latency, the main limiting factor arises from the long prompt and model response that must be processed sequentially. The good news is that this too can be mitigated, through fast fine-tuning techniques (<xref ref-type="bibr" rid="B5">Han et al., 2024</xref>) which allow a foundational LLM to be modified such that smaller prompts yield the same output.</p>
<p>In terms of scalability regarding the number of slices, as the number of slices increases, the effectiveness of learning-based policies would be affected because both the state space and the action space increase accordingly. This leads to longer training periods and convergence issues. The maximum number of slices would be limited by the available Resource Units (RUs) in the deployed system. In Wi-Fi 6 (802.11ax), the theoretical maximum number of RUs available in a 160&#xa0;MHz channel is 996. However, typical deployments often use smaller channel bandwidths due to regulatory constraints and spectrum availability. For a 20&#xa0;MHz channel, there are up to 9 RUs available. On the technical implementation side, if the User Priority (UP) field of Wi-Fi is used to map packets to slices, the number of slices would be limited to 8. For the LLM policy, as the Wi-Fi network scales in complexity, e.g., with more slices, the prompt fed to the LLM would have an impact on the latency of the response, as the &#x201d;reasoning&#x201d; of the LLM would describe more computations or consequences for each of the managed resources. The exact scaling factor will depend on the conditioning prompt.</p>
</sec>
<sec id="s4-2">
<title>4.2 Impact of state augmentation in REINFORCE</title>
<p>In addition, it is worth noting that the results show a more significant improvement from integrating the state-augmented approach into the REINFORCE algorithm (SAC-RE) in the more challenging scenario. As shown in <xref ref-type="fig" rid="F9">Figure 9</xref>, SAC-RE appears on the second Pareto front in the random walk scenario, dominating the vanilla RE approach without state-augmentation. In contrast, in the periodic scenario (<xref ref-type="fig" rid="F8">Figure 8</xref>), RE is co-located with SAC-RE on the second Pareto front, albeit with a significant trade-off in latency.</p>
</sec>
<sec id="s4-3">
<title>4.3 Conclusion</title>
<p>In conclusion, we confirmed the benefit of adding state augmentation to reinforcement learning solutions for systems exhibiting multi-modality, as the implemented SAC-RE solution outperformed the other RL approaches under a similarly low-cost, manually tuned hyper-parameter settings. The most promising result is the performance of the LLM-based policy, which required no training or optimization and, out-of-the-box, provided a dominating solution that can be further guided through prompt design. As LLM optimization techniques continue to evolve, it should become increasingly feasible to deploy LLM-based policies in low-cost scenarios as well.</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>Publicly available source code was analyzed in this study. This data can be found here: <ext-link ext-link-type="uri" xlink:href="https://gitlab.netcom.it.uc3m.es/predict-6g/AI-based_Wi-Fi_Slicing">https://gitlab.netcom.it.uc3m.es/predict-6g/AI-based_Wi-Fi_Slicing</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>RR: Conceptualization, Investigation, Methodology, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. DC: Funding acquisition, Supervision, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work has been partially funded by the European Commission Horizon Europe SNS JU PREDICT-6G (GA 101095890) Project.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>Authors RR and DC were employed by Intel Corporation.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. Generative AI was used to correct grammar of human writing.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Acharya</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Velasquez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>H. H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A survey on symbolic knowledge distillation of large language models</article-title>. <source>IEEE Trans. Artif. Intell.</source> <volume>5</volume>, <fpage>5928</fpage>&#x2013;<lpage>5948</lpage>. <pub-id pub-id-type="doi">10.1109/TAI.2024.3428519</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brockman</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cheung</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Pettersson</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Openai gym</article-title>. <comment>
<italic>arXiv preprint arXiv:1606.01540</italic>
</comment>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calvo-Fullana</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Paternain</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chamon</surname>
<given-names>L. F. O.</given-names>
</name>
<name>
<surname>Ribeiro</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>State augmented constrained reinforcement learning: overcoming the limitations of learning with rewards</article-title>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Candell</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Montgomery</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hany</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Sudhakaran</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Albrecht</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Cavalcanti</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Operational impacts of IEEE 802.1qbv scheduling on a collaborative robotic scenario</article-title>,&#x201d; in <source>Iecon 2022 - 48th annual Conference of the IEEE industrial electronics society</source> (<publisher-loc>Brussels, Belgium</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/IECON49645.2022.9968494</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S. Q.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Parameter-efficient fine-tuning for large models: a comprehensive survey</article-title>. <source>Corr. abs/2403</source>, <fpage>14608</fpage>. <pub-id pub-id-type="doi">10.48550/ARXIV.2403.14608</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Henderson</surname>
<given-names>T. R.</given-names>
</name>
<name>
<surname>Lacage</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Riley</surname>
<given-names>G. F.</given-names>
</name>
<name>
<surname>Dowell</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kopena</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2008</year>). <article-title>Network simulations with the ns-3 simulator</article-title>. <source>SIGCOMM Demonstr.</source> <volume>14</volume>, <fpage>527</fpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Adam: a method for stochastic optimization</article-title>,&#x201d; in <source>3rd international conference on learning representations, ICLR 2015</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<publisher-loc>San Diego, CA</publisher-loc>: <publisher-name>Conference Track Proceedings</publisher-name>).</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Onslicing: online end-to-end network slicing with reinforcement learning</article-title>,&#x201d; in <source>Proceedings of the 17th international conference on emerging networking EXperiments and technologies</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery</publisher-name>), <volume>21</volume>, <fpage>141</fpage>&#x2013;<lpage>153</lpage>. <pub-id pub-id-type="doi">10.1145/3485983.3494850</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>A constrained reinforcement learning based approach for network slicing</article-title>,&#x201d; in <source>2020 IEEE 28th international conference on network protocols (ICNP)</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. <pub-id pub-id-type="doi">10.1109/ICNP49622.2020.9259378</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Llm-pruner: on the structural pruning of large language models</article-title>,&#x201d; in <source>Advances in neural information processing systems 36: annual conference on neural information processing systems 2023</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Oh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Naumann</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Globerson</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Saenko</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hardt</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<publisher-loc>New Orleans, LA</publisher-loc>: <publisher-name>NeurIPS</publisher-name>).</citation>
</ref>
<ref id="B11">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mnih</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Badia</surname>
<given-names>A. P.</given-names>
</name>
<name>
<surname>Mirza</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lillicrap</surname>
<given-names>T. P.</given-names>
</name>
<name>
<surname>Harley</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Asynchronous methods for deep reinforcement learning</article-title>,&#x201d; in <source>Proceedings of the 33nd international conference on machine learning, ICML 2016</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Balcan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Weinberger</surname>
<given-names>K. Q.</given-names>
</name>
</person-group> (<publisher-loc>New York City, NY</publisher-loc>: <publisher-name>JMLR.org</publisher-name>), <volume>48</volume>, <fpage>1928</fpage>&#x2013;<lpage>1937</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>NaderiAlizadeh</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Eisen</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ribeiro</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>State-augmented learnable algorithms for resource management in wireless networks</article-title>. <source>IEEE Trans. Signal Process.</source> <volume>70</volume>, <fpage>5898</fpage>&#x2013;<lpage>5912</lpage>. <pub-id pub-id-type="doi">10.1109/TSP.2022.3229948</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Conceptual reinforcement learning for language-conditioned tasks</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>37</volume>, <fpage>9426</fpage>&#x2013;<lpage>9434</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v37i8.26129</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Rosales</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Python implementation of SAC-RE algorithm</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://gitlab.netcom.it.uc3m.es/predict-6g/AI-based_Wi-Fi_Slicing">https://gitlab.netcom.it.uc3m.es/predict-6g/AI-based_Wi-Fi_Slicing</ext-link> (Accessed March 14, 2025)</comment>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wolski</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klimov</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Proximal policy optimization algorithms</article-title>. <source>Corr. abs/1707</source>, <fpage>06347</fpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="web">
<collab>Stable-Baselines3</collab> (<year>2025a</year>). <article-title>Proximal policy optimization algorithm</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://stable-baselines3.readthedocs.io/en/master/modules/ppo.html">https://stable-baselines3.readthedocs.io/en/master/modules/ppo.html</ext-link> (Accessed March 14, 2025)</comment>.</citation>
</ref>
<ref id="B17">
<citation citation-type="web">
<collab>Stable-Baselines3</collab> (<year>2025b</year>). <article-title>Synchronous, deterministic variant of asynchronous advantage actor critic (A3C)</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://stable-baselines3.readthedocs.io/en/master/modules/a2c.html">https://stable-baselines3.readthedocs.io/en/master/modules/a2c.html</ext-link> (Accessed March 14, 2025)</comment>.</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sutton</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Barto</surname>
<given-names>A. G.</given-names>
</name>
</person-group> (<year>1998</year>). <source>Reinforcement learning: an introduction</source>, <volume>1</volume>. <publisher-name>MIT press Cambridge</publisher-name>.</citation>
</ref>
<ref id="B19">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Uslu</surname>
<given-names>Y. B.</given-names>
</name>
<name>
<surname>Doostnejad</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ribeiro</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>NaderiAlizadeh</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Learning to slice wi-fi networks: a state-augmented primal-dual approach</article-title>,&#x201d; in <source>Globecom 2024 - 2024 IEEE global communications conference</source>, <fpage>4521</fpage>&#x2013;<lpage>4527</lpage>. <pub-id pub-id-type="doi">10.1109/GLOBECOM52923.2024.10901174</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Williams</surname>
<given-names>R. J.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>Simple statistical gradient-following algorithms for connectionist reinforcement learning</article-title>. <source>Mach. Learn.</source> <volume>8</volume>, <fpage>229</fpage>&#x2013;<lpage>256</lpage>. <pub-id pub-id-type="doi">10.1007/BF00992696</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>S.-P.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sydir</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Advancing ran slicing with offline reinforcement learning</article-title>,&#x201d; in <source>2024 IEEE international symposium on dynamic spectrum access networks (DySPAN)</source>, <fpage>331</fpage>&#x2013;<lpage>338</lpage>. <pub-id pub-id-type="doi">10.1109/DySPAN60163.2024.10632750</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). &#x201c;<article-title>ns3-ai: fostering artificial intelligence algorithms for networking research</article-title>,&#x201d; in <source>Proceedings of the 2020 workshop on ns-3, WNS3 2020</source>. Editors <person-group person-group-type="editor">
<name>
<surname>Rouil</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Avallone</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Coudron</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gamess</surname>
<given-names>E.</given-names>
</name>
</person-group> (<publisher-loc>Gaithersburg, MD</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>57</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.1145/3389400.3389404</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zangooei</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Saha</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Golkarifard</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Boutaba</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Reinforcement learning for radio resource management in ran slicing: a survey</article-title>. <source>IEEE Commun. Mag.</source> <volume>61</volume>, <fpage>118</fpage>&#x2013;<lpage>124</lpage>. <pub-id pub-id-type="doi">10.1109/MCOM.004.2200532</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>An overview of network slicing for 5g</article-title>. <source>IEEE Wirel. Commun.</source> <volume>26</volume>, <fpage>111</fpage>&#x2013;<lpage>117</lpage>. <pub-id pub-id-type="doi">10.1109/mwc.2019.1800234</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Small</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Inverse reinforcement learning with natural language goals</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>35</volume>, <fpage>11116</fpage>&#x2013;<lpage>11124</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v35i12.17326</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>