<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" 'JATS-journalpublishing1-3-mathml3.dtd'>
<article article-type="research-article" dtd-version="1.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title-group>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1625968</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1625968</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Adaptive mapless mobile robot navigation using deep reinforcement learning based improved TD3 algorithm</article-title>
<alt-title alt-title-type="left-running-head">Nasti et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1625968">10.3389/frobt.2025.1625968</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Nasti</surname>
<given-names>Shoaib Mohd</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3060203"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal Analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing&#x2013;review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing&#x2013;original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Najar</surname>
<given-names>Zahoor Ahmad</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3230709"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing&#x2013;review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chishti</surname>
<given-names>Mohammad Ahsan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1988421"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &#x26; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/">Writing&#x2013;review and editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
</contrib>
</contrib-group>
<aff id="aff1">
<label>1</label>
<institution>Department of Information Technology, Central University of Kashmir</institution>, <city>Ganderbal</city>, <state>Jammu and Kashmir</state>, <country country="IN">India</country>
</aff>
<aff id="aff2">
<label>2</label>
<institution>Department of Computer Science and Engineering, National Institute of Technology</institution>, <city>Srinagar</city>, <state>Jammu and Kashmir</state>, <country country="IN">India</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Shoaib Mohd Nasti, <email xlink:href="mailto:shoaibnasti@cukashmir.ac.in">shoaibnasti@cukashmir.ac.in</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-12-18">
<day>18</day>
<month>12</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1625968</elocation-id>
<history>
<date date-type="received">
<day>14</day>
<month>05</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>08</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>11</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Nasti, Najar and Chishti.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Nasti, Najar and Chishti</copyright-holder>
<license>
<ali:license_ref start_date="2025-12-18">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>Navigating in unknown environments without prior maps poses a significant challenge for mobile robots due to sparse rewards, dynamic obstacles, and limited prior knowledge. This paper presents an Improved Deep Reinforcement Learning (DRL) framework based on the Twin Delayed Deep Deterministic Policy Gradient (TD3) algorithm for adaptive mapless navigation. In addition to architectural enhancements, the proposed method offers theoretical benefits byincorporates a latent-state encoder and predictor module to transform high-dimensional sensor inputs into compact embeddings. This compact representation reduces the effective dimensionality of the state space, enabling smoother value-function approximation and mitigating overestimation errors common in actor&#x2013;critic methods. It uses intrinsic rewards derived from prediction error in the latent space to promote exploration of novel states. The intrinsic reward encourages the agent to prioritize uncertain yet informative regions, improving exploration efficiency under sparse extrinsic reward signals and accelerating convergence. Furthermore, training stability is achieved through regularization of the latent space via maximum mean discrepancy (MMD) loss. By enforcing consistent latent dynamics, the MMD constraint reduces variance in target value estimation and results in more stable policy updates. Experimental results in simulated ROS2/Gazebo environments demonstrate that the proposed framework outperforms standard TD3 and other improved TD3 variants. Our model achieves a 93.1% success rate and a low 6.8% collision rate, reflecting efficient and safe goal-directed navigation. These findings confirm that combining intrinsic motivation, structured representation learning, and regularization-based stabilization produces more robust and generalizable policies for mapless mobile robot navigation.</p>
</abstract>
<kwd-group>
<kwd>adaptive navigation</kwd>
<kwd>deep reinforcement learning</kwd>
<kwd>mapless navigation</kwd>
<kwd>mobile robot</kwd>
<kwd>twin delayed DDPG</kwd>
</kwd-group>
<funding-group>
<funding-statement>The authors declare that no financial support was received for the research and/or publication of this article.</funding-statement>
</funding-group>
<counts>
<fig-count count="8"/>
<table-count count="4"/>
<equation-count count="11"/>
<ref-count count="29"/>
<page-count count="14"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Robot Learning and Evolution</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>Deep reinforcement learning (DRL) has emerged as a powerful framework for control and decision-making in robotics, enabling end-to-end learning of complex navigation policies without explicit programming. RL methods such as Deep Q-Networks (DQN) and Deep Deterministic Policy Gradient (DDPG) have demonstrated success in video games and continuous control tasks (<xref ref-type="bibr" rid="B11">Lillicrap et al., 2015</xref>). In robotics, DRL promises to overcome the limitations of classical planners by learning directly from sensor observations and environmental interactions (<xref ref-type="bibr" rid="B9">Kober and Peters, 2013</xref>; <xref ref-type="bibr" rid="B23">Tang et al., 2024</xref>). In particular, mapless navigation, steering a mobile robot to a goal without <italic>a priori</italic> maps, is an active research area due to its importance for deployment in unknown or dynamic environments where mapping is difficult or costly. Traditional approaches rely on explicit mapping and path planning algorithms (e.g., SLAM followed by A&#x2a; or DWA) (<xref ref-type="bibr" rid="B1">Fox et al., 1997</xref>; <xref ref-type="bibr" rid="B8">Khatib, 1986</xref>), but these can fail in cluttered or partially observable settings and require manual tuning. DRL can potentially learn robust local navigation behaviors directly from sensor inputs, adapting to unseen obstacles (<xref ref-type="bibr" rid="B22">Tai et al., 2017</xref>). However, DRL for mapless navigation faces challenges such as sparse rewards, sample inefficiency, and safety constraints (<xref ref-type="bibr" rid="B10">Li et al., 2023</xref>).</p>
<p>To address these, we develop an Improved Twin Delayed DDPG (Improved TD3) algorithm that augments the standard TD3 architecture with several enhancements: a learned latent state representation via an autoencoder-style encoder, an auxiliary predictor model that estimates the next latent state given the current state and action, and intrinsic rewards based on prediction error. Our method is motivated by recent work on curiosity-driven exploration and representation learning in RL (<xref ref-type="bibr" rid="B17">Pathak et al., 2017</xref>), and the need for efficient exploration in mapless navigation (<xref ref-type="bibr" rid="B20">Ruan et al., 2022</xref>). In summary, our contributions are.<list list-type="bullet">
<list-item>
<p>A comprehensive extension of the TD3 algorithm for mapless mobile robot navigation, incorporating a latent encoder-predictor and intrinsic reward that guides exploration.</p>
</list-item>
<list-item>
<p>A detailed implementation including mathematical definitions of the encoder, critic and actor networks, and intrinsic reward computation.</p>
</list-item>
<list-item>
<p>Experimental validation in ROS2/Gazebo simulation showing that the improved TD3 significantly outperforms the standard TD3 baseline in mean reward, success rate, and collision avoidance.</p>
</list-item>
</list>
</p>
<p>The rest of the paper is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> reviews related work; <xref ref-type="sec" rid="s3">Section 3</xref> presents the standard TD3 algorithm; <xref ref-type="sec" rid="s4">Section 4</xref> details the improved TD3 method; <xref ref-type="sec" rid="s5">Section 5</xref> describes experiments, metrics, and results; and <xref ref-type="sec" rid="s6">Section 6</xref> concludes with a summary and future directions.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<sec id="s2-1">
<label>2.1</label>
<title>DRL for navigation</title>
<p>Reinforcement learning has been increasingly applied to robotic navigation tasks. Early RL-based navigation used discrete controllers or small state spaces, but modern DRL uses neural networks to handle high-dimensional inputs (e.g., images or LIDAR). For instance, <xref ref-type="bibr" rid="B22">Tai et al. (2017)</xref> demonstrated a mapless navigation planner trained end-to-end with asynchronous DDPG using a sparse 10-dimensional laser scan and relative target position; their planner transferred from simulation to a real robot without explicit mapping.</p>
<p>More recent works address exploration and sample efficiency in mapless navigation. For example, <xref ref-type="bibr" rid="B24">Yadav et al. (2023)</xref> employed Soft Actor-Critic (SAC) with curriculum learning and dual prioritized replay for mobile robot navigation, highlighting the challenge of sparse rewards. <xref ref-type="bibr" rid="B20">Ruan et al. (2022)</xref> introduced a curiosity-based intrinsic motivation combined with a temporal cognition module for visual navigation, suggesting that self-generated rewards can improve exploration. <xref ref-type="bibr" rid="B10">Li et al. (2023)</xref> combined a human-designed gap-detection planner with a DRL agent, effectively incorporating prior knowledge to accelerate learning in complex scenes. These approaches emphasize the value of intrinsic rewards or hybrid methods to overcome DRL limitations in navigation.</p>
<p>Parallel to DRL, classical navigation methods remain widely used. Techniques such as potential fields (<xref ref-type="bibr" rid="B8">Khatib, 1986</xref>) and the Dynamic Window Approach (DWA) (<xref ref-type="bibr" rid="B1">Fox et al., 1997</xref>) perform reactive obstacle avoidance given a map or sensor data, but often require careful design and can get stuck in local minima. Hybrid methods have been explored (<xref ref-type="bibr" rid="B13">Liu et al., 2024</xref>): proposed TD3-DWA, combining TD3 with DWA by treating DWA parameters as tunable by the policy. Our work avoids the use of traditional planners such as DWA or A&#x2a;, relying instead on learning-based policies guided by goal-relative information and onboard sensing.</p>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>TD3 and actor-critic methods</title>
<p>Our baseline algorithm is Twin Delayed DDPG (TD3) (<xref ref-type="bibr" rid="B2">Fujimoto et al., 2018</xref>), an off-policy actor-critic method for continuous control. TD3 addresses function-approximation error in DDPG (<xref ref-type="bibr" rid="B11">Lillicrap et al., 2016</xref>) by three tricks: (1) using two independent Q-networks and taking the minimum for target computation (clipped double-Q learning); (2) adding clipped noise to the target policy (policy smoothing regularization); and (3) delaying actor updates relative to critic updates. TD3 typically outperforms vanilla DDPG and some on-policy methods due to reduced overestimation and variance. Several works have since extended or optimized TD3 for robotics.</p>
<p>Actor-critic methods like TD3, DDPG, and SAC (<xref ref-type="bibr" rid="B4">Haarnoja et al., 2018</xref>) have been favored in robotics for handling continuous actions and sample reuse via replay. SAC adds entropy maximization for exploration stability (<xref ref-type="bibr" rid="B4">Haarnoja et al., 2018</xref>). Asynchronous variants like A3C/A2C (<xref ref-type="bibr" rid="B15">Mnih et al., 2016</xref>) also apply to navigation, but rely on on-policy updates. In our work we build on TD3 due to its strong performance and off-policy efficiency, while incorporating additional modules to tackle exploration in unknown environments.</p>
</sec>
<sec id="s2-3">
<label>2.3</label>
<title>Intrinsic motivation and representation learning</title>
<p>Intrinsic reward strategies have proven effective in improving exploration when extrinsic rewards are sparse. A common approach is to train a predictor network and use its error as a curiosity signal (<xref ref-type="bibr" rid="B17">Pathak et al., 2017</xref>). For instance, <xref ref-type="bibr" rid="B17">Pathak et al. (2017)</xref> defined curiosity as the error in predicting the consequence of actions in a learned feature space, encouraging exploration of novel states. Others have used prediction uncertainty or state-counts for intrinsic rewards. In line with this, our Improved TD3 uses the error of a learned state-transition predictor (operating in latent space) as an intrinsic reward, thus motivating the agent to visit states where its model is poor.</p>
</sec>
<sec id="s2-4">
<label>2.4</label>
<title>Improved TD3 variants</title>
<p>Several recent studies have extended the TD3 framework for mobile robot navigation. <xref ref-type="bibr" rid="B6">Jeng and Chiang (2023)</xref> introduced survival-based penalty shaping in TD3 to improve convergence and reduce collision. <xref ref-type="bibr" rid="B25">Yang et al. (2024)</xref> added an intrinsic curiosity module (ICM) and randomness-enhancement to help the agent escape sparse-reward local optima. <xref ref-type="bibr" rid="B7">Kashyap and Konathalapalli (2025)</xref> compared TD3 against DDPG and DQN on a TurtleBot3 platform using ROS2 and LiDAR, showing that TD3 offered the best performance.</p>
<p>
<xref ref-type="bibr" rid="B16">Neamah and Mayorga Mayorga (2024)</xref> proposed Optimized TD3 (O-TD3) with prioritized replay and parallel critic updates, achieving high success in human-crowded scenarios. <xref ref-type="bibr" rid="B18">Raj and Kos (2024)</xref> implemented dynamic delay updates, Ornstein&#x2013;Uhlenbeck noise, and curriculum transfer learning on top of TD3. <xref ref-type="bibr" rid="B5">Huang et al. (2024)</xref> developed LP-TD3, which fuses LSTM memory, prioritized experience replay, and intrinsic curiosity rewards, significantly improving convergence and generalization.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Standard TD3 algorithm</title>
<p>We first review the standard TD3 algorithm (as in <xref ref-type="bibr" rid="B2">Fujimoto et al. (2018)</xref>) to establish notation. We consider a Markov decision process with continuous state space <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi mathvariant="script">S</mml:mi>
<mml:mo>&#x2286;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and action space <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi mathvariant="script">A</mml:mi>
<mml:mo>&#x2286;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. At each time step <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the agent observes state <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, selects action <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> given the deterministic policy (actor) <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, receives reward <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and next state <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The goal is to maximize the discounted return <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>
<xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref> summarizes standard TD3. Here the actor and critic are typically multilayer neural networks with hidden layers of size 256, ReLU activations, and a final <inline-formula id="inf41">
<mml:math id="m41">
<mml:mrow>
<mml:mi>tanh</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>on the actor output (to enforce action bounds). TD3&#x2019;s three heuristics (min of two Q&#x2019;s, target action noise, and delayed updates) together reduce overestimation and variance. In practice, we set the policy noise std <inline-formula id="inf42">
<mml:math id="m42">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>noise</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, clip <inline-formula id="inf43">
<mml:math id="m43">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, policy delay <inline-formula id="inf44">
<mml:math id="m44">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, and soft-update rate <inline-formula id="inf45">
<mml:math id="m45">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.005</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>as in <xref ref-type="bibr" rid="B2">Fujimoto et al. (2018)</xref>. These settings are also used in our improved <xref ref-type="statement" rid="Algorithm_2">Algorithm 2</xref> for fair comparison.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>Standard TD3.</p>
<p>
<list list-type="simple">
<list-item>
<p>1:&#x2003;Initialize actor <inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, critics <inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and their target networks <inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>with same weights</p>
</list-item>
<list-item>
<p>2:&#x2003;Initialize replay buffer <inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:mi mathvariant="script">B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>3:&#x2003;<bold>for</bold>each training step <inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>4:&#x2003;&#x2003;&#x2003;Observe state <inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, select action</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
<inline-formula id="inf173">
<mml:math id="m184">
<mml:mrow>
<mml:mspace width="1em"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mtext>with&#x2009;exploration&#x2009;noise</mml:mtext>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>5:&#x2003;&#x2003;&#x2003;Execute <inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, observe reward <inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>and next state <inline-formula id="inf29">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>6:&#x2003;&#x2003;&#x2003;Store transition <inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>into <inline-formula id="inf31">
<mml:math id="m31">
<mml:mrow>
<mml:mi mathvariant="script">B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>7:&#x2003;&#x2003;&#x2003;Sample a mini-batch of transitions from <inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:mi mathvariant="script">B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>and compute target</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="normal">c</mml:mi>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mi mathvariant="normal">p</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
<inline-formula id="inf174">
<mml:math id="m185">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>noise</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf34">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
<inline-formula id="inf175">
<mml:math id="m186">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>8:&#x2003;&#x2003;&#x2003;Update each critic <inline-formula id="inf35">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>by minimizing the loss</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf36">
<mml:math id="m36">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>9:&#x2003;&#x2003;&#x2003;<bold>if</bold>every <inline-formula id="inf37">
<mml:math id="m37">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>steps (policy delay) <bold>then</bold>
</p>
</list-item>
<list-item>
<p>10:&#x2003;&#x2003;&#x2003;&#x2003;Update actor using sampled policy gradient:</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf38">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>J</mml:mi>
<mml:mo>&#x2248;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>11:&#x2003;&#x2003;&#x2003;&#x2003;Perform a gradient step to maximize <inline-formula id="inf39">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>:</p>
</list-item>
<list-item>
<p>12:&#x2003;&#x2003;&#x2003;&#x2003;Soft update target networks</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf40">
<mml:math id="m40">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>&#x3d5;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>13:&#x2003;&#x2003;<bold>end</bold> <bold>if</bold>
</p>
</list-item>
<list-item>
<p>14:&#x2003;<bold>end</bold> <bold>for</bold>
</p>
</list-item>
</list>
</p>
</statement>
</p>
<p>
<statement content-type="algorithm" id="Algorithm_2">
<label>Algorithm 2</label>
<p>Improved TD3 (ITD3).</p>
<p>
<list list-type="simple">
<list-item>
<p>&#x2003;&#x2003;Initialize encoder <inline-formula id="inf111">
<mml:math id="m121">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, critics <inline-formula id="inf112">
<mml:math id="m122">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, actor <inline-formula id="inf113">
<mml:math id="m123">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>2&#x2003;Initialize targets: <inline-formula id="inf114">
<mml:math id="m124">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf115">
<mml:math id="m125">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2190;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf116">
<mml:math id="m126">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Initialize replay buffer <inline-formula id="inf117">
<mml:math id="m127">
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>; set <inline-formula id="inf118">
<mml:math id="m128">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>4:&#x2003;<bold>for</bold> episode &#x3d; <inline-formula id="inf119">
<mml:math id="m129">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>episodes</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;Reset environment state <inline-formula id="inf120">
<mml:math id="m130">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>6:&#x2003;&#x2003;&#x2003;<bold>for</bold> <inline-formula id="inf121">
<mml:math id="m131">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf122">
<mml:math id="m132">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>8:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf123">
<mml:math id="m133">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;Execute <inline-formula id="inf124">
<mml:math id="m134">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, observe <inline-formula id="inf125">
<mml:math id="m135">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>10:&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf126">
<mml:math id="m136">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf127">
<mml:math id="m137">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>Predictor</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>12:&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf128">
<mml:math id="m138">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf129">
<mml:math id="m139">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>14:&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf130">
<mml:math id="m140">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>int</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf131">
<mml:math id="m141">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>int</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>16:&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf132">
<mml:math id="m142">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>in buffer <inline-formula id="inf133">
<mml:math id="m143">
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf134">
<mml:math id="m144">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2190;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>18:&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf135">
<mml:math id="m145">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>from <inline-formula id="inf136">
<mml:math id="m146">
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf137">
<mml:math id="m147">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>20:&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf138">
<mml:math id="m148">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mtext>clip</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;Compute targets:</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf139">
<mml:math id="m149">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
<inline-formula id="inf176">
<mml:math id="m187">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>22:&#x2003;&#x2003;&#x2003;&#x2003;Update critics</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf140">
<mml:math id="m150">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
<inline-formula id="inf177">
<mml:math id="m188">
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;Compute MMD loss:</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf141">
<mml:math id="m151">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>MMD</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mtext>MMD</mml:mtext>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>24:&#x2003;&#x2003;&#x2003;&#x2003;Update encoder <inline-formula id="inf142">
<mml:math id="m152">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>:</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf143">
<mml:math id="m153">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>enc</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>MMD</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>MMD</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<bold>if</bold>&#x2003;<inline-formula id="inf144">
<mml:math id="m154">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>mod</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>delay</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> <bold>then</bold>
</p>
</list-item>
<list-item>
<p>26:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;Update actor:</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf145">
<mml:math id="m155">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo movablelimits="false" form="prefix">&#x2211;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;Soft update targets:</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<inline-formula id="inf146">
<mml:math id="m156">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>
<inline-formula id="inf178">
<mml:math id="m189">
<mml:mrow>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>&#x3d5;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>&#x3c8;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>28:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<bold>end</bold> <bold>if</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<bold>if</bold> <inline-formula id="inf147">
<mml:math id="m157">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>is True <bold>then</bold>
</p>
</list-item>
<list-item>
<p>30:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<bold>break</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<bold>end</bold> <bold>if</bold>
</p>
</list-item>
<list-item>
<p>32:&#x2003;&#x2003;&#x2003;<bold>end</bold> <bold>for</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;<bold>end</bold> <bold>for</bold>
</p>
</list-item>
</list>
</p>
</statement>
</p>
<p>The TD3 critic consists of two Q-networks <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> with parameters <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and corresponding target networks <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The actor (policy) network is <inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> with parameters <inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and has a delayed target <inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The algorithm proceeds as follows.</p>
</sec>
<sec id="s4">
<label>4</label>
<title>Proposed improved TD3 (ITD3)</title>
<p>Our Proposed Improved TD3 (ITD3) augments the standard architecture with an encoder&#x2013;predictor module, an intrinsic reward signal, and latent space regularization via MMD. We describe each component and the overall training procedure in detail.</p>
<sec id="s4-1">
<label>4.1</label>
<title>Encoder&#x2013;predictor architecture</title>
<p>We introduce a learned latent encoding of the state to capture useful features. Let the original state vector be <inline-formula id="inf46">
<mml:math id="m46">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> (e.g., LIDAR readings plus robot pose). An encoder network <inline-formula id="inf47">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> produces a latent embedding <inline-formula id="inf48">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> (we use <inline-formula id="inf49">
<mml:math id="m49">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). Concretely, <inline-formula id="inf50">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a feedforward network with two hidden layers of size <inline-formula id="inf51">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>enc</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and nonlinear activations (e.g., ReLU), followed by an AvgL1Norm layer normalizing the output. Formally:<disp-formula id="equ1">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">A</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
<mml:mi mathvariant="normal">g</mml:mi>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mn mathvariant="normal">1</mml:mn>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mi mathvariant="normal">r</mml:mi>
<mml:mi mathvariant="normal">m</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf52">
<mml:math id="m53">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the ReLU activation, <inline-formula id="inf53">
<mml:math id="m54">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are weight matrices, and AvgL1Norm normalizes the vector.</p>
<p>Given <inline-formula id="inf54">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and an action <inline-formula id="inf55">
<mml:math id="m56">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, we also define a predictor network that estimates the next latent state. Specifically, we compute <inline-formula id="inf56">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> via two hidden layers of size 256 and ReLU activation:<disp-formula id="equ2">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>&#x3c3;</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mspace width="0.13em"/>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>with input <inline-formula id="inf57">
<mml:math id="m59">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> concatenation. This predictor shares weights with the encoder beyond the first layer, ensuring <inline-formula id="inf58">
<mml:math id="m60">
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf59">
<mml:math id="m61">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> use the same latent basis. The next latent of the actual next state is <inline-formula id="inf60">
<mml:math id="m62">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>The actor (policy) network is modified to condition on the latent: instead of taking only <inline-formula id="inf61">
<mml:math id="m63">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, it first encodes <inline-formula id="inf62">
<mml:math id="m64">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf63">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and then computes the action. In practice we implement a split architecture: the actor first processes <inline-formula id="inf64">
<mml:math id="m66">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> through a hidden layer, concatenates the resulting features with <inline-formula id="inf65">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, then applies two more layers and a <inline-formula id="inf66">
<mml:math id="m68">
<mml:mrow>
<mml:mi>tanh</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> output layer. The critic networks likewise take <inline-formula id="inf67">
<mml:math id="m69">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, but also incorporate <inline-formula id="inf68">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf69">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> as additional inputs: each Q-network processes <inline-formula id="inf70">
<mml:math id="m72">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> alongside <inline-formula id="inf71">
<mml:math id="m73">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> (with appropriate layers). This way, the critic value depends on both the raw state-action and the learned latent embedding.</p>
<sec id="s4-1-1">
<label>4.1.1</label>
<title>Benefits of the encoder&#x2013;predictor</title>
<p>TD3 can directly process continuous states; however, our encoder&#x2013;predictor compresses high-dimensional sensor streams (e.g., hundreds of LiDAR beams plus pose) into a compact latent vector. Representation-learning shows that such compact, structured representations improve sample efficiency and performance by simplifying the critic&#x2019;s function-approximation task and avoiding overfitting to noisy inputs. The predictor further supplies an auxiliary self-supervised signal that organizes latent features according to dynamics, encouraging better generalization. In our experiments, training curves (see <xref ref-type="sec" rid="s5">Section 5</xref>) reveal that ITD3 attains high evaluation rewards earlier than a baseline TD3 operating directly on raw states, and exhibits lower variance in Q-values and losses, highlighting improved sample efficiency and stability. Moreover, the latent embedding dimension is set to <inline-formula id="inf72">
<mml:math id="m74">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> to balance information preservation and simplicity; too large latent space slows learning, whereas too small a space loses essential information A very large latent dimension risks redundancy and overfitting&#x2014;adding parameters without improving performance and potentially slowing convergence&#x2014;while a very small dimension can discard essential features, leading to information loss and suboptimal policies. We chose <inline-formula id="inf73">
<mml:math id="m75">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> as a practical balance: it substantially compresses the high-dimensional LiDAR and pose observations but still provides sufficient capacity for capturing salient structure. During preliminary hyperparameter tuning, smaller latent dimensions (e.g., 128) resulted in reduced final success rates and poorer generalization, while larger dimensions (e.g., 512) offered no noticeable benefit and increased training time. Thus <inline-formula id="inf74">
<mml:math id="m76">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> serves as a reasonable trade-off between information retention and simplicity, consistent with state representation learning guidelines This design therefore offers tangible benefits without introducing unnecessary redundancy.</p>
</sec>
<sec id="s4-1-2">
<label>4.1.2</label>
<title>Weight-sharing rationale and expressive capacity</title>
<p>Although the predictor shares weights with the encoder after the first hidden layer, this design choice does not unduly limit expressiveness. Weight sharing, also known as weight tying, is a well-established regularization technique in neural networks: by reducing the number of distinct parameters, it saves memory and helps prevent overfitting. In language models, weight tying between embedding and softmax layers reduces parameter count and improves generalization. Similarly, sharing the encoder&#x2019;s second-layer weights <inline-formula id="inf75">
<mml:math id="m77">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> ensures that the encoded latent <inline-formula id="inf76">
<mml:math id="m78">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and the predicted latent <inline-formula id="inf77">
<mml:math id="m79">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> live in the same feature space, aligning the predictor&#x2019;s output with the encoder&#x2019;s representation and stabilizing the intrinsic reward signal. The predictor still has its own final layer, so it retains sufficient expressive power to model the state-transition dynamics.</p>
</sec>
</sec>
<sec id="s4-2">
<label>4.2</label>
<title>Intrinsic reward via prediction error</title>
<p>We incorporate an intrinsic reward <inline-formula id="inf78">
<mml:math id="m80">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>int</mml:mtext>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> to encourage exploration of states that are hard to predict. After sampling a transition <inline-formula id="inf79">
<mml:math id="m81">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> from the replay buffer, we compute the latent encodings <inline-formula id="inf80">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf81">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and the predicted next latent <inline-formula id="inf82">
<mml:math id="m84">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>We define the raw prediction error as<disp-formula id="equ3">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>Let the running maximum be<disp-formula id="equ4">
<mml:math id="m86">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="2em"/>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
<p>We set <inline-formula id="inf83">
<mml:math id="m87">
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>8</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The normalized intrinsic reward is<disp-formula id="equ5">
<mml:math id="m88">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>int</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>and the total reward used for TD target computation is<disp-formula id="equ6">
<mml:math id="m89">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>ext</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>int</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mspace width="2em"/>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.1</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf84">
<mml:math id="m90">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>ext</mml:mtext>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is the environment reward. To prevent intrinsic bonuses from destabilizing learning when the predictor is initially inaccurate, we employ two controls. First, we scale the bonus by a small constant <inline-formula id="inf85">
<mml:math id="m91">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, keeping it modest relative to extrinsic rewards. Second, we normalize the prediction error <inline-formula id="inf86">
<mml:math id="m92">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> by a running maximum <inline-formula id="inf87">
<mml:math id="m93">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> with <inline-formula id="inf88">
<mml:math id="m94">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The normalized curiosity <inline-formula id="inf89">
<mml:math id="m95">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>int</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> bounds the intrinsic term <inline-formula id="inf90">
<mml:math id="m96">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>int</mml:mtext>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> within <inline-formula id="inf91">
<mml:math id="m97">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, avoiding large spikes early in training and preventing the intrinsic component from dominating the learning signal.</p>
</sec>
<sec id="s4-3">
<label>4.3</label>
<title>Encoder training and disentanglement loss</title>
<p>While the intrinsic reward shapes exploration, the encoder&#x2013;predictor must itself be learned. At each training step, we update the encoder parameters <inline-formula id="inf92">
<mml:math id="m98">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to minimize the prediction error. Specifically, we compute the loss for a sampled batch:<disp-formula id="equ7">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>enc</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>MMD</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf93">
<mml:math id="m100">
<mml:mrow>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mi mathvariant="normal">D</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the maximum mean discrepancy between the distribution of encoded states <inline-formula id="inf94">
<mml:math id="m101">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and predicted states <inline-formula id="inf95">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (using a Gaussian kernel) (<xref ref-type="bibr" rid="B3">Gretton et al., 2012</xref>). We use <inline-formula id="inf96">
<mml:math id="m103">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>MMD</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1.0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The MMD term acts as a disentanglement or regularization loss: it encourages the latent distribution of <inline-formula id="inf97">
<mml:math id="m104">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> transitions to match the marginal latent distribution of states, promoting consistency. In practice, the code computes MMD by sampling pairwise kernel differences. The encoder loss is then backpropagated and the encoder parameters <inline-formula id="inf98">
<mml:math id="m105">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> updated by Adam. We use a learning rate of <inline-formula id="inf99">
<mml:math id="m106">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for the encoder.</p>
</sec>
<sec id="s4-4">
<label>4.4</label>
<title>Actor and critic updates</title>
<p>After computing the intrinsic reward and updating the encoder, we update the critics. We first form the target Q-value using the combined reward:<disp-formula id="equ8">
<mml:math id="m107">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf100">
<mml:math id="m108">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.99</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> is the discount factor. The critic networks <inline-formula id="inf101">
<mml:math id="m109">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are then trained to regress to <inline-formula id="inf102">
<mml:math id="m110">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. We use standard MSE loss and update <inline-formula id="inf103">
<mml:math id="m111">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> via Adam with learning rate <inline-formula id="inf104">
<mml:math id="m112">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>We update the actor every <inline-formula id="inf105">
<mml:math id="m113">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>y</mml:mi>
<mml:mtext>_</mml:mtext>
<mml:mi>f</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> critic updates. Specifically, we compute the actor loss as the negative Q-value under the current critic:<disp-formula id="equ9">
<mml:math id="m114">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>actor</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf106">
<mml:math id="m115">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf107">
<mml:math id="m116">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for on-policy action. This encourages the policy to choose actions with high predicted Q-value. We take a gradient step on <inline-formula id="inf108">
<mml:math id="m117">
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> with learning rate <inline-formula id="inf109">
<mml:math id="m118">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>We use soft updates for the target networks every training step:<disp-formula id="equ10">
<mml:math id="m119">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2190;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:mi>&#x3d5;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>with <inline-formula id="inf110">
<mml:math id="m120">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.005</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> as is standard in TD3.</p>
<p>
<xref ref-type="statement" rid="Algorithm_2">Algorithm 2</xref> provides the high-level training loop. For clarity, we summarize in words.<list list-type="bullet">
<list-item>
<p>Intrinsic/extrinsic reward: Each transition yields extrinsic reward plus a scaled curiosity bonus based on normalized latent prediction error.</p>
</list-item>
<list-item>
<p>Encoder/predictor update: Minimize mean-squared error between predicted and actual next latent, plus MMD regularization.</p>
</list-item>
<list-item>
<p>Critic update: Compute TD-target using minimum of two target critics and total reward; minimize standard MSE loss.</p>
</list-item>
<list-item>
<p>Actor update: Every few steps, maximize the Q-value by gradient ascent on the first critic; update actor target network.</p>
</list-item>
</list>
</p>
<sec id="s4-4-1">
<label>4.4.1</label>
<title>Training stability</title>
<p>We examined whether joint updates of the encoder&#x2013;predictor, critics, and actor could introduce instability. These modules are optimized with distinct losses and disjoint parameter sets, which mitigates direct gradient interference. To further ensure stability&#x2014;especially when the predictor is inaccurate early in training&#x2014;we bound the intrinsic term by normalizing the latent prediction error with a running maximum and scaling it by a small coefficient <inline-formula id="inf148">
<mml:math id="m158">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, thereby constraining it to <inline-formula id="inf149">
<mml:math id="m159">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Standard TD3 stabilizers (policy delay <inline-formula id="inf150">
<mml:math id="m160">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, target policy noise with clipping, soft target updates) and MMD regularization also damp abrupt changes. Throughout training, critic losses, actor gradient norms, and encoder losses remained well-behaved with no spikes or divergence.</p>
</sec>
<sec id="s4-4-2">
<label>4.4.2</label>
<title>Multi-input critic</title>
<p>Our critic receives <inline-formula id="inf151">
<mml:math id="m161">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> as input, which increases the number of input features compared to a standard TD3 critic. This design does not lead to overfitting for several reasons. First, the latent vectors <inline-formula id="inf152">
<mml:math id="m162">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf153">
<mml:math id="m163">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are compact (256-dimensional) summaries of the high-dimensional state; they reduce the input space. Second, the latent space is regularized via the MMD loss, which encourages smoothness and discourages spurious correlations. Third, weight sharing acts as a form of regularization by reducing the number of trainable parameters. Our training and evaluation curves show no signs of overfitting: the critic&#x2019;s loss remains stable and the evaluation performance tracks the training performance closely. Thus, the multi-input critic benefits from richer information without sacrificing generalization.</p>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Experiments and results</title>
<p>We evaluate our Improved TD3 (ITD3) approach on a simulated indoor navigation task. The setup uses ROS2 Humble and Gazebo 11 Classic on Ubuntu 22.04, with NVIDIA Quadro P4000 GPU (NVIDIA), Intel&#xae; Xeon(R) CPU E5-1650 v4 @ 3.60 GHz <inline-formula id="inf154">
<mml:math id="m164">
<mml:mrow>
<mml:mo>&#xd7;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 12 CPU and 64 GB RAM. The agent is a differential-drive mobile robot equipped with a 2D LIDAR. The state vector <inline-formula id="inf155">
<mml:math id="m165">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> includes the LIDAR distance measurements (truncated to a fixed range) and the agent&#x2019;s relative goal position and orientation. The action space consists of continuous linear and angular velocities (bounded by <inline-formula id="inf156">
<mml:math id="m166">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> m/s and <inline-formula id="inf157">
<mml:math id="m167">
<mml:mrow>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> rad/s).</p>
<p>We train the ITD3 model using the structured procedure detailed in <xref ref-type="statement" rid="Algorithm_2">Algorithm 2</xref>. The goal in each episode is a fixed target location within the environment with randomly placed obstacles (see <xref ref-type="fig" rid="F5">Figure 5</xref>). Episodes are capped at 500 steps. We report results averaged over 100 test episodes after training, measuring success rate, collision rate, and time to reach the goal. The hyperparameters are given in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Hyperparameters and system details.</p>
</caption>
<table>
<tbody valign="top">
<tr>
<td align="left">Discount factor <inline-formula id="inf158">
<mml:math id="m168">
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">0.99</td>
</tr>
<tr>
<td align="left">Batch size</td>
<td align="left">128</td>
</tr>
<tr>
<td align="left">Replay buffer size</td>
<td align="left">
<inline-formula id="inf159">
<mml:math id="m169">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Target update rate</td>
<td align="left">250 steps</td>
</tr>
<tr>
<td align="left">Exploration noise start/end</td>
<td align="left">1.0 <inline-formula id="inf160">
<mml:math id="m170">
<mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> 0.1 over 500k steps</td>
</tr>
<tr>
<td align="left">Policy delay</td>
<td align="left">2</td>
</tr>
<tr>
<td align="left">Target policy noise</td>
<td align="left">0.2</td>
</tr>
<tr>
<td align="left">Noise clip</td>
<td align="left">0.5</td>
</tr>
<tr>
<td align="left">MMD regularization weight <inline-formula id="inf161">
<mml:math id="m171">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>MMD</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">1.0</td>
</tr>
<tr>
<td align="left">Intrinsic reward weight <inline-formula id="inf162">
<mml:math id="m172">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">0.1</td>
</tr>
<tr>
<td align="left">Intrinsic normalization</td>
<td align="left">Running max of <inline-formula id="inf163">
<mml:math id="m173">
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, with <inline-formula id="inf164">
<mml:math id="m174">
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>8</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Encoder dim <inline-formula id="inf165">
<mml:math id="m175">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">256</td>
</tr>
<tr>
<td align="left">Hidden dims</td>
<td align="left">256 (each network)</td>
</tr>
<tr>
<td align="left">Optimizer</td>
<td align="left">Adam</td>
</tr>
<tr>
<td align="left">Learning rates (actor, critic, encoder)</td>
<td align="left">3e-4</td>
</tr>
<tr>
<td align="left">System</td>
<td align="left">Ubuntu 22.04, ROS2 Humble, Gazebo 11, P4000 GPU, 64 GB RAM</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s5-1">
<label>5.1</label>
<title>Evaluation metrics</title>
<p>During training, we periodically evaluate the current policy (without exploration noise) over several episodes to measure performance. We plot the mean total reward per evaluation epoch to track learning progress. After training, we report: (1) Success Rate: fraction of episodes reaching the goal; (2) Collision Rate: fraction of episodes ending in collision (Timeouts are treated as failures and counted under collisions); (3) Average Time: mean steps to reach goal (for successes).</p>
</sec>
<sec id="s5-2">
<label>5.2</label>
<title>Quantitative results</title>
<p>
<xref ref-type="fig" rid="F1">Figure 1</xref> shows the evaluation rewards over training epochs. The shaded regions show individual episode rewards, and the solid lines are running averages. We see that Improved TD3 quickly attains higher rewards and stabilizes around a larger mean. This indicates that the intrinsic rewards and encoding help the agent learn a more effective navigation policy.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Mean evaluation total reward. Running mean (window &#x3d; 10) is shown as thicker lines.</p>
</caption>
<graphic xlink:href="frobt-12-1625968-g001.tif">
<alt-text content-type="machine-generated">Line graph titled &#x22;Evaluation Rewards Over 100 Epochs (10 episodes/epoch)&#x22; showing mean total reward across evaluation epochs. The blue line represents the mean total reward, with sharp fluctuations. The orange line shows the running mean total reward with a window size of ten, depicting a smoother trend. The y-axis ranges from negative fifty to one hundred fifty, and the x-axis ranges from zero to two hundred epochs.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s5-3">
<label>5.3</label>
<title>Training curve analysis</title>
<p>To assess the learning behavior of the proposed Improved TD3 agent, we analyzed several key training metrics using TensorBoard visualizations. Comparing ITD3 with a baseline TD3 (without the encoder&#x2013;predictor) confirms these advantages: ITD3&#x2019;s evaluation reward curve rises more quickly and stabilizes with a narrower confidence band, indicating faster learning and reduced variance. This aligns with the expectation that compact latent representations facilitate sample-efficient value approximation. These include the Q-values, target Q-values, Q-max, loss, and their corresponding target counterparts.</p>
<sec id="s5-3-1">
<label>5.3.1</label>
<title>Q and target Q</title>
<p>The Q-values represent the expected return estimated by the critic networks for the current policy. As shown in <xref ref-type="fig" rid="F2">Figure 2</xref>, the Q-values initially drop due to random policy actions, then gradually increase, stabilizing around the <inline-formula id="inf166">
<mml:math id="m176">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>15</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> mark. This improvement indicates that the critic successfully learns to evaluate more rewarding state-action pairs. The Target Q-values, generated by the target critic networks, follow a similar trend but with slightly smoother progression. This aligns with their role in providing stable targets during training updates, essential for avoiding overestimation bias and ensuring training stability.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Evolution of Q-values from the current critic network (left) and the target critic network (right) during training.</p>
</caption>
<graphic xlink:href="frobt-12-1625968-g002.tif">
<alt-text content-type="machine-generated">Two line graphs show the data labeled &#x22;Q&#x22; over time, each with a different color. The left graph uses a dark line and shows values fluctuating between approximately -40 and -10. The right graph, with an orange line, displays similar fluctuations over the same range. Both graphs show a steep decline at the start before stabilizing with consistent oscillations. Each x-axis ranges from 0 to 1 million, with y-axis values from -40 to -10.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s5-3-2">
<label>5.3.2</label>
<title>Q<sub>max</sub> and target Q<sub>max</sub>
</title>
<p>These plots capture the maximum predicted Q-values across all actions for a given state. In <xref ref-type="fig" rid="F3">Figure 3</xref>, Q<sub>max</sub> shows sharp increases at around 100k and 450k steps, indicating sudden improvements in the agent&#x2019;s policy that allows for higher return expectations. The stability in Q<sub>max</sub> beyond these points suggests convergence in learning optimal actions. Similarly, the target Q<sub>max</sub> curve displays delayed but parallel improvement trends, which reinforces the critic&#x2019;s evolving confidence in the best action-value estimations.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>
<inline-formula id="inf167">
<mml:math id="m177">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> predicted by the critic (left) and target critic (right) networks.</p>
</caption>
<graphic xlink:href="frobt-12-1625968-g003.tif">
<alt-text content-type="machine-generated">Two line graphs are displayed, both titled &#x22;Q_max.&#x22; The left graph has a black line, and the right graph has an orange line. Both graphs show a step-like increase from approximately 100 to over 200 on the y-axis, with the x-axis ranging from zero to one million.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s5-3-3">
<label>5.3.3</label>
<title>Critic loss</title>
<p>The loss curve in <xref ref-type="fig" rid="F4">Figure 4</xref> reveals a high variance in early training (0&#x2013;300k steps), corresponding to unstable critic predictions due to random exploration. Gradual smoothing and reduction of loss beyond 500k steps reflect more accurate value function approximations as the critic aligns with the target Q-values. A lower and stable loss near the end of training is indicative of convergence and reduced estimation error.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Critic loss showing the mean squared error between predicted and target Q-values.</p>
</caption>
<graphic xlink:href="frobt-12-1625968-g004.tif">
<alt-text content-type="machine-generated">Line graph showing loss over iterations from zero to one million. Initially, loss decreases sharply from above eleven to below seven, then fluctuates around six with slight downward trend.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s5-3-4">
<label>5.3.4</label>
<title>Summary of interpretations</title>
<p>
<list list-type="bullet">
<list-item>
<p>Q-values (Critic): Indicate growing understanding of long-term returns; increasing and stabilizing over time.</p>
</list-item>
<list-item>
<p>Target Q-values: Act as a stable guide for updating the critic; follow similar trend but smoother.</p>
</list-item>
<list-item>
<p>Q<sub>max</sub>: Reflects peaks in learned optimal policies; abrupt rises correlate with performance breakthroughs.</p>
</list-item>
<list-item>
<p>Target Q<sub>max</sub>: Lags Q<sub>max</sub> slightly but confirms learning trend.</p>
</list-item>
<list-item>
<p>Loss: Decreasing loss validates effective convergence of the critic networks.</p>
</list-item>
</list>
</p>
<p>These metrics collectively validate that the improved TD3 model exhibits stable and progressive learning behavior, effectively minimizing critic errors and maximizing action-value predictions over training steps.</p>
</sec>
</sec>
<sec id="s5-4">
<label>5.4</label>
<title>Evaluation of improved TD3 performance</title>
<p>To evaluate the effectiveness of our Improved TD3 (ITD3) framework, we conducted extensive testing. The metrics used include success rate, collision rate and average time to goal as shown in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Performance metrics of Improved TD3 over 100 test episodes. All episodes terminate either upon success or collision; timeouts are treated as failures and counted under collisions.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Metric</th>
<th align="center">Improved TD3</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Success Rate</td>
<td align="center">0.931</td>
</tr>
<tr>
<td align="left">Collision Rate</td>
<td align="center">0.068</td>
</tr>
<tr>
<td align="left">Average Time</td>
<td align="center">12.91</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The Improved TD3 agent achieves a success rate of 93.14%, indicating reliable goal-reaching behavior across diverse and previously unseen environments. The collision rate is as low as 6.84%, showing that the agent learns to avoid obstacles effectively. Moreover, the average number of steps to reach the goal is reduced to 12.91, suggesting faster convergence and efficient navigation.</p>
<p>These improvements stem from our enhancements to the standard TD3 framework, including latent state encoding, intrinsic curiosity-driven rewards, and MMD-based regularization. Together, they enable the robot to explore more intelligently, learn more efficiently, and generalize better across different scenarios.</p>
</sec>
<sec id="s5-5">
<label>5.5</label>
<title>Qualitative analysis</title>
<p>
<xref ref-type="fig" rid="F5">Figure 5</xref> shows the top view of the Gazebo Simulation Environment, in which the black circle under the spotlight is the robot, and the white circles are the obstacles. The walls and the surroundings also act as obstacles. <xref ref-type="fig" rid="F6">Figure 6</xref> is the rviz of the same Gazebo environment depicted in <xref ref-type="fig" rid="F5">Figure 5</xref>. Here, the red arcs represent the LIDAR scan of the robot. We can visualise the presence of obstacles through this laser scan, as shown by the green curves in <xref ref-type="fig" rid="F6">Figure 6</xref>. The sensor data indicates four obstacles clearly, as shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, while the remaining two are already accounted for but not directly visible, as they are positioned behind the other two. (In total, there are six obstacles, but due to their alignment, two are hidden behind others, which is why the robot only senses four directly).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Top view of the Gazebo simulation environment.</p>
</caption>
<graphic xlink:href="frobt-12-1625968-g005.tif">
<alt-text content-type="machine-generated">Top-down view of a grid-based maze with walls forming an enclosure and two internal corridors. A central object emits blue light beams, casting shadows from white oval-shaped objects placed within the grid.</alt-text>
</graphic>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Corresponding Rviz view of the Gazebo simulation environment shown in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
</caption>
<graphic xlink:href="frobt-12-1625968-g006.tif">
<alt-text content-type="machine-generated">Grid-based simulation illustrating wave propagation from a central black source. Red concentric circles represent waves moving towards green and magenta elements on the right. A color scale from red to blue is at the top.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="F7">Figure 7</xref> demonstrates that the robot successfully moved toward the goal location, marked by the green dot, while avoiding all obstacles. The red line represents the path taken by the robot. The curves along this path indicate the presence of obstacles that the robot avoided while navigating successfully toward the goal. The other red lines depict the previous trajectories that the robot took. A detailed observation of the laser scan reveals that the robot has effectively sensed the obstacles within its range, including the walls and surrounding structures. <xref ref-type="fig" rid="F8">Figure 8</xref> demonstrates that the robot has successfully evaded all the obstacles in its path while moving towards the goal location. The policy learned to drive the Robot forward while avoiding collisions. The presence of intrinsic reward during training helped the agent learn to maintain broad coverage with its sensors, encouraging diverse exploration and helping avoid local traps such as corners or walls. Across many trials, the Improved TD3 robot consistently navigates effectively. These qualitative observations align with the quantitative gains: improved exploration leads to faster and safer goal-reaching.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Rviz visualization showing the robot&#x2019;s navigation trajectory.</p>
</caption>
<graphic xlink:href="frobt-12-1625968-g007.tif">
<alt-text content-type="machine-generated">A gray grid displays red dotted lines forming curved and straight paths, with a colorful section in the bottom left corner featuring concentric arcs and a black circle. A red horizontal bar appears on the right.</alt-text>
</graphic>
</fig>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Gazebo environment during navigation phase.</p>
</caption>
<graphic xlink:href="frobt-12-1625968-g008.tif">
<alt-text content-type="machine-generated">Diagram showing a top-down view of a room with wooden walls and several white cylindrical objects inside. A blue triangular area represents a cone of vision from a camera positioned near one wall, highlighting its field of view.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s5-6">
<label>5.6</label>
<title>Benchmark comparison</title>
<p>We evaluated our Improved TD3 (ITD3) against the state of the art by comparing key metrics reported in related work, including training efficiency, success and collision rates and cumulative reward. For example, <xref ref-type="bibr" rid="B6">Jeng and Chiang (2023)</xref> introduced a survival-penalty reward shaping in an end-to-end TD3/DDPG framework. They report that TD3 converges faster and more stably than DDPG, yielding a higher task success rate in evaluation. In their parking and maze scenarios, TD3 achieved a markedly higher success rate while keeping collisions low, demonstrating the benefit of their collision-penalty reward shaping.</p>
<p>
<xref ref-type="bibr" rid="B25">Yang et al. (2024)</xref> focus on intrinsic rewards. By integrating an Intrinsic Curiosity Module (ICM) and a Randomness-Enhanced Module (REM) with TD3, they report that their ICM &#x2b; REM-TD3 method achieved an 83.5% success rate out of 1000 test episodes&#x2013;significantly higher than baseline DRL methods&#x2013;with only 3 episodes exceeding the time limit. This corresponds to a collision rate of only about 16.2% (compared to 23.1% for vanilla TD3 in their <xref ref-type="table" rid="T1">Table 1</xref>), and a substantial reduction in average episode length. These results indicate that the intrinsic reward mechanism accelerated learning and exploration, improving success and reducing collisions. (<xref ref-type="bibr" rid="B25">Yang et al., 2024</xref>). also report the performance of an A3C baseline, which achieved a 62.5% success rate with a 32.4% collision rate in their <xref ref-type="table" rid="T1">Table 1</xref>. Compared to TD3 and TD3-based intrinsic-reward variants, A3C shows noticeably weaker navigation and poorer obstacle-avoidance capability, indicating limited exploration efficiency under sparse-reward conditions.</p>
<p>
<xref ref-type="bibr" rid="B7">Kashyap and Konathalapalli (2025)</xref> compare TD3, DDPG, and DQN on a TurtleBot3 platform in static and dynamic obstacle courses. They report that TD3 yields the most efficient navigation overall. Quantitatively, TD3 reduced travel time by roughly 14%&#x2013;27% relative to DDPG and 28%&#x2013;55% relative to DQN, while also cutting collision counts per episode by similar factors. In other words, TD3 achieved the highest success rates and average rewards across diverse scenarios, leveraging ROS2 and LiDAR integration for robust perception. The combination of high success and low collisions in these tests underscores the robustness of TD3 under more complex, sensor-rich settings.</p>
<p>
<xref ref-type="bibr" rid="B16">Neamah and Mayorga Mayorga (2024)</xref> propose an optimized TD3 tuned for crowded human&#x2013;robot interaction. They report exceptionally high performance: after training for 12,000 episodes, their method achieved a 92% success rate in evaluation, requiring on average only 772 steps (about 11 s) to reach each target. By contrast, their baseline DQN only achieved 64% success. These results highlight that a well-tuned TD3 can learn very efficient, collision-free navigation policies even in highly dynamic human-populated environments.</p>
<p>In addition to TD3-based methods, other reinforcement learning algorithms have also been applied to mapless navigation. On-policy methods such as A3C (<xref ref-type="bibr" rid="B15">Mnih et al., 2016</xref>) and PPO (<xref ref-type="bibr" rid="B21">Schulman et al., 2017</xref>) learn directly from raw sensor data and have demonstrated success in similar ROS-based environments, but typically require more training samples and careful reward shaping. For instance, in a ROS 2 &#x2b; Gazebo setup with LiDAR inputs, a PPO-trained TurtleBot3 achieved an 82% success rate after extensive training (<xref ref-type="bibr" rid="B21">Schulman et al., 2017</xref>), Similarly, A3C has been used for end-to-end navigation (<xref ref-type="bibr" rid="B14">Mirowski et al., 2017</xref>), but its sample-inefficient on-policy updates often result in moderate performance unless combined with auxiliary objectives or curricula. Off-policy Soft Actor-Critic (SAC) (<xref ref-type="bibr" rid="B4">Haarnoja et al., 2018</xref>) tends to be more sample-efficient; however, SAC alone can struggle with sparse rewards in navigation. Recent extensions like SAC-LSTM (<xref ref-type="bibr" rid="B26">Zhang and Chen, 2023</xref>) (incorporating memory) report success rates of 89% in highly dynamic scenarios), highlighting that augmenting SAC with memory and exploration modules improves performance. These comparisons suggest that our ITD3&#x2019;s combination of intrinsic motivation and representation learning yields competitive or better performance than A3C, PPO, and SAC in similar mapless navigation tasks, with faster convergence and higher success rates.</p>
<p>Other recent works likewise report strong performance. For instance, <xref ref-type="bibr" rid="B18">Raj and Kos (2024)</xref> used a DQN-based DRL controller and demonstrated high target-reaching success rates and reward gains compared to PPO on complex mazes. <xref ref-type="bibr" rid="B5">Huang et al. (2024)</xref> apply an improved TD3 with LSTM and curiosity modules to mapless inspection tasks, reporting performance improvements (e.g., higher success and reward) over prior baselines (they cite &#x201c;curiosity-driven&#x201d; exploration). Taken together, these benchmarks show that our Improved TD3 (ITD3) consistently matches or exceeds the success and efficiency of recent mapless navigation methods: all report success rates in the 70%&#x2013;90% range with low collision rates, and our results fall at the high end of this range. In parallel to navigation, related perception-driven architectures have improved robustness in thermal or cross-modal settings. Deep-IRTarget leverages dual-domain spatial and frequency cues to enhance thermal target localization (<xref ref-type="bibr" rid="B27">Zhang et al., 2022</xref>), while DFANet preserves modality-specific IR/RGB information before attention-based fusion, yielding more reliable downstream representations (<xref ref-type="bibr" rid="B28">Zhang et al., 2024</xref>). Further, cognition-driven structural-prior modeling has been shown effective for instance-dependent label noise, where STMN aligns transition matrices with human-plausible structural constraints to suppress invalid label transitions (<xref ref-type="bibr" rid="B29">Zhang et al., 2025</xref>). These findings reinforce the broader value of structured representation learning and prior-guided regularization, which conceptually motivate our use of compact latent encoding with regularization and intrinsic learning signals. In addition to scalar metrics, we note key architectural enhancements across these works. Many use advanced reward shaping: for example, <xref ref-type="bibr" rid="B6">Jeng and Chiang (2023)</xref> employ a &#x201c;survival penalty&#x201d; function to combat sparse rewards, enabling collision-free paths. <xref ref-type="bibr" rid="B25">Yang et al. (2024)</xref> integrate a curiosity-driven ICM and a randomness-enhancement (REM) module to provide dense intrinsic rewards and encourage exploration. Although the encoder&#x2013;predictor adds modest complexity, the benefits of compressing the observation space (improved sample efficiency and stability) and the auxiliary predictive signal justify its inclusion. Representation-learning suggests that an optimal latent space should be low-dimensional yet informative; our architecture adheres to this principle. The latent dimension <inline-formula id="inf168">
<mml:math id="m178">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> was selected to balance information retention and computational efficiency. According to state representation learning principles, the representation space should be constrained to a low dimensionality while remaining sufficiently informative. Dimensions that are too high risk redundancy and overfitting, whereas overly compressed spaces omit critical features. Our experiments indicated that <inline-formula id="inf169">
<mml:math id="m179">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>256</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> achieves this balance for the sensor modalities considered. <xref ref-type="bibr" rid="B7">Kashyap and Konathalapalli (2025)</xref> uses middleware (ROS2) and LiDAR sensors to improve perception and robustness. <xref ref-type="bibr" rid="B5">Huang et al. (2024)</xref> augment TD3 with recurrent memory (LSTM) and curiosity-based rewards. Collectively, these innovations (survival penalties, intrinsic rewards, sensor fusion, memory models) are aimed at improving convergence and final policy quality. In our design, we incorporate an adaptive reward scheme to stabilize learning. By comparison with these, our method achieves better training efficiency and navigation performance.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s6">
<label>6</label>
<title>Discussion</title>
<p>The results and comparisons presented in <xref ref-type="table" rid="T3">Table 3</xref> highlight key trends in the development of TD3-based mapless navigation strategies. A major insight is that incremental algorithmic enhancements, including intrinsic motivation, reward shaping, and memory-augmented architectures, play a decisive role in improving both learning efficiency and final policy quality.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Benchmark comparison of TD3-based navigation methods.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Method (Reference)</th>
<th align="center">Technique/Key features</th>
<th align="center">Success rate (%)</th>
<th align="center">Collision rate (%)</th>
<th align="center">Remarks/Notes</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B6">Jeng and Chiang (2023)</xref>
</td>
<td align="center">Survival-penalty reward shaping with TD3 in 270&#xb0; Scenario</td>
<td align="center">&#x2dc;92</td>
<td align="center">8</td>
<td align="center">Fast convergence, higher success vs. DDPG</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B25">Yang et al. (2024)</xref>
</td>
<td align="center">ICM &#x2b; Randomness-enhanced TD3</td>
<td align="center">83.5</td>
<td align="center">16.5</td>
<td align="center">0.3% timeouts out of 1000 trials, better exploration</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B25">Yang et al. (2024)</xref>
</td>
<td align="center">A3C Baseline</td>
<td align="center">62.5</td>
<td align="center">37.5</td>
<td align="center">Lower success and higher collision vs. TD3</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B7">Kashyap and Konathalapalli (2025)</xref>
</td>
<td align="center">Standard TD3 with ROS2, LiDAR</td>
<td align="center">91</td>
<td align="center">9</td>
<td align="center">TD3 outperforms DDPG/DQN</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B16">Neamah and Mayorga Mayorga (2024)</xref>
</td>
<td align="center">Optimized TD3 (parallel updates, prioritized replay)</td>
<td align="center">92</td>
<td align="center">8</td>
<td align="center">Effective in human-populated settings</td>
</tr>
<tr>
<td align="left">PPO (<xref ref-type="bibr" rid="B21">Schulman et al., 2017</xref>)</td>
<td align="center">On-policy PPO with LiDAR inputs</td>
<td align="center">82</td>
<td align="center">18</td>
<td align="center">ROS2/Gazebo experiment (<xref ref-type="bibr" rid="B19">Rana and Kaveendran, 2025</xref>); good reliability but slower convergence than TD3.</td>
</tr>
<tr>
<td align="left">SAC-LSTM (<xref ref-type="bibr" rid="B26">Zhang and Chen, 2023</xref>)</td>
<td align="center">Off-policy SAC with LSTM memory</td>
<td align="center">89</td>
<td align="center">11</td>
<td align="center">Achieved 89% in dynamic scenarios; memory boosts performance.</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B5">Huang et al. (2024)</xref>
</td>
<td align="center">LP-TD3: LSTM &#x2b; PER &#x2b; ICM</td>
<td align="center">Not reported</td>
<td align="center">Not reported</td>
<td align="center">Strong gains in exploration and convergence</td>
</tr>
<tr>
<td align="left">
<bold>ITD3 (Our)</bold>
</td>
<td align="center">Improved TD3 &#x2b; intrinsic rewards &#x2b; MMD</td>
<td align="center">
<bold>93.1</bold>
</td>
<td align="center">
<bold>6.8</bold>
</td>
<td align="center">Fast, stable learning with hybrid rewards and encoder</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>In several benchmark studies, only the success rate is explicitly reported, while the collision rate is often omitted. In such cases, we adopt a reasonable assumption&#x2014;commonly implied in navigation literature&#x2014;that the sum of success and failure rates (including collisions and timeouts) approximates <inline-formula id="inf170">
<mml:math id="m180">
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Therefore, wherever not directly provided, the collision rate is computed as the complement of the reported success rate. This estimation ensures consistent cross-method comparison. Bold values indicate the best performing metric(s) among the methods compared. Success rate is the percentage of trials that reached the goal within the episode limit. Collision rate is the percentage of trials that ended in a collision.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Among these, intrinsic reward mechanisms such as those used by <xref ref-type="bibr" rid="B25">Yang et al. (2024)</xref> (ICM and REM) and in our Improved TD3 (ITD3) method stand out for their ability to tackle the sparse reward problem. ITD3&#x2019;s intrinsic module, driven by prediction error in latent space, offers dense learning signals that encourage broader exploration. This aligns with <xref ref-type="bibr" rid="B25">Yang et al. (2024)</xref>&#x2019;s findings, where intrinsic feedback significantly improved convergence and reduced timeouts. Unlike their setup, however, our approach integrates latent encoding and MMD regularization, producing more structured exploration.</p>
<p>In addition, <xref ref-type="bibr" rid="B6">Jeng and Chiang (2023)</xref> and <xref ref-type="bibr" rid="B16">Neamah and Mayorga Mayorga (2024)</xref> underscore the value of reward shaping and hyperparameter tuning. Their success rates (92%) are comparable to ours, but they rely either on custom-designed penalties or extensive critic optimization. In contrast, our method achieves a slightly higher success rate (93.1%) by combining architectural modularity (encoder, predictor, curiosity) with latent space regularization via MMD, minimizing the need for aggressive tuning.</p>
<p>From a robotics perspective, <xref ref-type="bibr" rid="B7">Kashyap and Konathalapalli (2025)</xref> demonstrate the practical applicability of standard TD3 using real-world middleware (ROS2) and perception modules (LiDAR), which validates its feasibility in deployed systems. Our method builds on this by incorporating mapless adaptability and reward-driven learning, making it more flexible for unstructured environments.</p>
<p>The synthesis of the benchmarking results reveals three critical patterns.<list list-type="bullet">
<list-item>
<p>Curiosity-driven learning is central: All methods with intrinsic components (ITD3, <xref ref-type="bibr" rid="B25">Yang et al. (2024)</xref>, <xref ref-type="bibr" rid="B5">Huang et al. (2024)</xref>) report noticeable gains in exploration and success.</p>
</list-item>
<list-item>
<p>Representation learning improves generalization: ITD3&#x2019;s use of latent space encoding and MMD regularization helps the agent generalize better to novel situations.</p>
</list-item>
<list-item>
<p>Modular architectures perform better: ITD3&#x2019;s combination of encoder-predictor and intrinsic rewards shows that thoughtfully combining components (rather than just tuning TD3) can improve robustness.</p>
</list-item>
</list>
</p>
<p>These findings validate our proposed approach as a competitive and extensible framework for adaptive robot navigation in real-world settings.</p>
</sec>
<sec id="s7">
<label>7</label>
<title>Conclusion and future scope</title>
<p>We have proposed an improved mapless mobile robot navigation algorithm for unknown environments using an enhanced Twin Delayed Deep Deterministic Policy Gradient (TD3) framework. Our contributions aimed at addressing key limitations in standard DRL-based navigation&#x2014;namely, sparse extrinsic rewards and limited exploration in dynamic, cluttered environments.</p>
<p>The proposed algorithm integrates three critical components: (i) a latent-state encoder&#x2013;predictor module to abstract high-dimensional sensor inputs into compact, informative embeddings; (ii) an intrinsic reward mechanism based on prediction error to guide exploration toward underrepresented states; and (iii) latent space regularization via maximum mean discrepancy (MMD) to promote consistent and disentangled representations. Together, these components enhance both sample efficiency and policy robustness.</p>
<p>Experimental evaluations in ROS2/Gazebo simulation environments demonstrate the effectiveness of our approach. Compared to the standard TD3 baseline and several recent TD3 variants, our Improved TD3 (ITD3) model achieved the highest recorded success rate of 93.1% while reducing the collision rate to 6.8%. These results confirm that the intrinsic rewards encourage diverse and meaningful exploration, while the latent representation facilitates generalization across unseen scenarios. Furthermore, training curves and critic metrics reveal improved convergence behavior, reduced variance, and stable Q-value estimation throughout training.</p>
<p>Our benchmarking against related works further validates the competitiveness of our method. While other studies have explored reward shaping, curiosity modules, or representation learning individually, our framework unifies these strategies within a modular architecture. The addition of MMD regularization to enforce latent consistency represents a novel combination that enhances training in stochastic environments.</p>
<p>Several promising avenues exist for future work.<list list-type="bullet">
<list-item>
<p>Real-world deployment: We plan to transfer the trained policies onto physical robots, addressing sim-to-real challenges using domain randomization, sensor noise modeling, and transfer learning techniques.</p>
</list-item>
<list-item>
<p>Scalability and hierarchical control: Extending the model to larger environments and multi-room layouts may benefit from hierarchical RL strategies or global&#x2013;local policy decomposition.</p>
</list-item>
<list-item>
<p>Multi-agent coordination: Adapting the Improved TD3 framework for cooperative multi-robot tasks can enable distributed exploration and shared learning.</p>
</list-item>
<list-item>
<p>Hybrid model-based extensions: Incorporating lightweight dynamics models or predictive forward models alongside the encoder may further improve data efficiency and planning capabilities.</p>
</list-item>
<list-item>
<p>Safety-aware adaptation: Integrating safety constraints, human-in-the-loop feedback, or formal verification mechanisms can enable deployment in safety-critical domains.</p>
</list-item>
<list-item>
<p>Multi-step Extensions: Our intrinsic reward currently relies on single-step prediction error, which encourages exploration by rewarding states where the immediate dynamics model is inaccurate. However, for complex or slowly evolving dynamics, one-step prediction may not capture longer-term uncertainties. A potential extension is to train the predictor to forecast multiple future latents <inline-formula id="inf171">
<mml:math id="m181">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and compute a multi-step curiosity bonus, defined for example, as</p>
</list-item>
</list>
<disp-formula id="equ11">
<mml:math id="m182">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>multi</mml:mtext>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>j</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>where <inline-formula id="inf172">
<mml:math id="m183">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> discounts errors at longer horizons. Such a signal would encourage the agent to explore areas where its dynamics model is inaccurate over a longer time span, potentially improving robustness and planning for complex tasks. Nevertheless, multi-step prediction is more challenging to learn due to compounding errors, and balancing its computational cost with performance gains is an open research question. We leave the implementation of multi-step curiosity and temporal-consistency constraints for future work and note that one-step prediction was sufficient to achieve strong performance in our navigation tasks.</p>
<p>This paper presents a robust and extensible solution to adaptive navigation in unstructured settings, bridging the gap between raw sensor-driven DRL and practical deployment readiness. The ITD3 algorithm we proposed serves as a strong foundation for future autonomous systems capable of real-time decision-making in unknown environments.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>SN: Methodology, Formal Analysis, Data curation, Software, Investigation, Conceptualization, Writing &#x2013; review and editing, Visualization, Writing &#x2013; original draft. ZN: Resources, Supervision, Writing &#x2013; review and editing, Validation. MC: Validation, Writing &#x2013; review and editing, Resources, Supervision.</p>
</sec>
<ack>
<p>We gratefully acknowledge that our improved TD3 algorithm implementation was adapted from the TD3 algorithm implementation in an open source repository by Ahmed Nurye (2024), available under the MIT License.</p>
</ack>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s12">
<title>Generative AI statement</title>
<p>The authors declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s13">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1888428/overview">Shaoming He</ext-link>, Beijing Institute of Technology, China</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1804319/overview">Du Xinwu</ext-link>, Henan University of Science and Technology, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1810064/overview">Ruiheng Zhang</ext-link>, Beijing Institute of Technology, China</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fox</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Thrun</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>The dynamic window approach to collision avoidance</article-title>. <source>IEEE Robotics and Automation</source> <volume>4</volume> (<issue>1</issue>), <fpage>23</fpage>&#x2013;<lpage>33</lpage>. <pub-id pub-id-type="doi">10.1109/100.580977</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Fujimoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>van Hoof</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Meger</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Addressing function approximation error in actor-critic methods</article-title>,&#x201d; in <source>Proceedings of the 35th international conference on machine learning (ICML)</source>, <volume>80</volume>, <fpage>1587</fpage>&#x2013;<lpage>1596</lpage>.</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gretton</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Borgwardt</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rasch</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sch&#xf6;lkopf</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Smola</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A kernel two-sample test</article-title>. <source>J. Mach. Learn. Res.</source> <volume>13</volume>, <fpage>723</fpage>&#x2013;<lpage>773</lpage>. </mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Haarnoja</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor</article-title>,&#x201d; in <source>Proceedings of the international conference on machine learning (ICML)</source>.</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Inspection robot navigation based on improved td3 algorithm</article-title>. <source>Sensors</source> <volume>24</volume> (<issue>8</issue>), <fpage>2525</fpage>. <pub-id pub-id-type="doi">10.3390/s24082525</pub-id>
<pub-id pub-id-type="pmid">38676143</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jeng</surname>
<given-names>S.-L.</given-names>
</name>
<name>
<surname>Chiang</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>End-to-end autonomous navigation based on deep reinforcement learning with a survival penalty function</article-title>. <source>Sensors</source> <volume>23</volume> (<issue>20</issue>), <fpage>8651</fpage>. <pub-id pub-id-type="doi">10.3390/s23208651</pub-id>
<pub-id pub-id-type="pmid">37896743</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kashyap</surname>
<given-names>A. K.</given-names>
</name>
<name>
<surname>Konathalapalli</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Autonomous navigation of ros2 based turtlebot3 in static and dynamic environments using intelligent approach</article-title>. <source>Int. J. Inf. Technol.</source> <pub-id pub-id-type="doi">10.1007/s41870-025-02500-5</pub-id>
</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khatib</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>1986</year>). <article-title>Real-time obstacle avoidance for manipulators and mobile robots</article-title>. <source>Int. J. Robotics Res.</source> <volume>5</volume> (<issue>1</issue>), <fpage>90</fpage>&#x2013;<lpage>98</lpage>. <pub-id pub-id-type="doi">10.1177/027836498600500106</pub-id>
</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kober</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Reinforcement learning in robotics: a survey</article-title>. <source>Int. J. Robotics Res.</source> <volume>32</volume> (<issue>11</issue>), <fpage>1238</fpage>&#x2013;<lpage>1274</lpage>. <pub-id pub-id-type="doi">10.1177/0278364913495721</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An efficient deep reinforcement learning algorithm for mapless navigation with gap-guided switching strategy</article-title>. <source>J. Intelligent and Robotic Syst.</source> <volume>108</volume> (<issue>43</issue>), <fpage>43</fpage>. <pub-id pub-id-type="doi">10.1007/s10846-023-01888-1</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lillicrap</surname>
<given-names>T. P.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Heess</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Erez</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tassa</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2015</year>). <article-title>Continuous control with deep reinforcement learning</article-title>. <source>arXiv Preprint arXiv:1509.02971</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1509.02971</pub-id>
</mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lillicrap</surname>
<given-names>T. P.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Pritzel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Heess</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Erez</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tassa</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Continuous control with deep reinforcement learning</article-title>,&#x201d; in <source>4th international conference on learning representations (ICLR)</source>. <source>arXiv:1509.02971</source>.</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Q.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Td3 based collision free motion planning for robot navigation</article-title>. <source>arXiv Preprint arXiv:2405.15460</source>, <fpage>247</fpage>&#x2013;<lpage>250</lpage>. <pub-id pub-id-type="doi">10.1109/cisce62493.2024.10653233</pub-id>
</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mirowski</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Grimes</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Malinowski</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Learning to navigate in complex environments</article-title>,&#x201d; in <source>International conference on learning representations</source> (<publisher-name>ICLR Workshop</publisher-name>). <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1611.03673">https://arxiv.org/abs/1611.03673</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Mnih</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Badia</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mirza</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lillicrap</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Harley</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). &#x201c;<article-title>Asynchronous methods for deep reinforcement learning</article-title>,&#x201d; in <source>Proceedings of the international conference on machine learning (ICML)</source>.</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Neamah</surname>
<given-names>H. A.</given-names>
</name>
<name>
<surname>Mayorga Mayorga</surname>
<given-names>O. A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Optimized td3 algorithm for robust autonomous navigation in crowded and dynamic human-interaction environments</article-title>. <source>Results Eng.</source> <volume>24</volume>, <fpage>102874</fpage>. <pub-id pub-id-type="doi">10.1016/j.rineng.2024.102874</pub-id>
</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Pathak</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Efros</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Darrell</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Curiosity-driven exploration by self-supervised prediction</article-title>,&#x201d; in <source>Proceedings of the international conference on machine learning (ICML)</source>.</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raj</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kos</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Intelligent mobile robot navigation in unknown and complex environment using reinforcement learning technique</article-title>. <source>Sci. Rep.</source> <volume>14</volume> (<issue>1</issue>), <fpage>22852</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-024-72857-3</pub-id>
<pub-id pub-id-type="pmid">39354023</pub-id>
</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rana</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kaveendran</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Deep reinforcement learning with PPO for autonomous Mobile robot navigation using ROS 2 framework</article-title>. <source>Int. J. Res. Appl. Sci. and Eng. Technol.</source> <volume>13</volume> (<issue>7</issue>), <fpage>2119</fpage>&#x2013;<lpage>2125</lpage>. <pub-id pub-id-type="doi">10.22214/ijraset.2025.73330</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ruan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A target-driven visual navigation method based on intrinsic motivation exploration and space topological cognition</article-title>. <source>Sci. Rep.</source> <volume>12</volume>, <fpage>3462</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-022-07264-7</pub-id>
<pub-id pub-id-type="pmid">35236878</pub-id>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wolski</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klimov</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Proximal policy optimization algorithms</article-title>. <source>CoRR</source>, <fpage>06347</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1707.06347</pub-id>
</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tai</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Paolo</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Virtual-to-real deep reinforcement learning: continuous control of mobile robots for mapless navigation</article-title>,&#x201d; in <conf-name>IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>.</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Abbatematteo</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chandra</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Mart&#xed;n-Mart&#xed;n</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Deep reinforcement learning for robotics: a survey of real-world successes</article-title>. <source>arXiv Preprint arXiv:2408</source>, <fpage>03539v1</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.2408.03539</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yadav</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Rudall</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bakr</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Hein</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>R&#xfc;ckert</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Deep reinforcement learning for mapless navigation of autonomous mobile robot</article-title>. <source>Int. J. Comput. Sci. Trends Comput. Commun. (IJCSTCC)</source>, <fpage>283</fpage>&#x2013;<lpage>288</lpage>. <pub-id pub-id-type="doi">10.1109/icstcc59206.2023.10308456</pub-id>
</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Mobile robot navigation based on intrinsic reward mechanism with td3 algorithm</article-title>. <source>Int. J. Adv. Robotic Syst.</source> <volume>21</volume> (<issue>5</issue>), <fpage>17298806241292893</fpage>. <pub-id pub-id-type="doi">10.1177/17298806241292893</pub-id>
</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Path planning of a mobile robot for a dynamic indoor environment based on an sac-lstm algorithm</article-title>. <source>Sensors</source> <volume>23</volume> (<issue>24</issue>), <fpage>9802</fpage>. <pub-id pub-id-type="doi">10.3390/s23249802</pub-id>
<pub-id pub-id-type="pmid">38139648</pub-id>
</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep-irtarget: an automatic target detector in infrared imagery using dual-domain feature extraction and allocation</article-title>. <source>IEEE Trans. Multimedia</source> <volume>24</volume>, <fpage>1735</fpage>&#x2013;<lpage>1749</lpage>. <pub-id pub-id-type="doi">10.1109/tmm.2021.3070138</pub-id>
</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Differential feature awareness network within antagonistic learning for infrared-visible object detection</article-title>. <source>IEEE Trans. Circuits Syst. Video Technol.</source> <volume>34</volume> (<issue>8</issue>), <fpage>6735</fpage>&#x2013;<lpage>6748</lpage>. <pub-id pub-id-type="doi">10.1109/tcsvt.2023.3289142</pub-id>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Cao</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Si</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Cognition-driven structural prior for instance-dependent label transition matrix estimation</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>36</volume> (<issue>2</issue>), <fpage>3730</fpage>&#x2013;<lpage>3743</lpage>. <pub-id pub-id-type="doi">10.1109/tnnls.2023.3347633</pub-id>
<pub-id pub-id-type="pmid">38190682</pub-id>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>