<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mar. Sci.</journal-id>
<journal-title>Frontiers in Marine Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mar. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-7745</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmars.2025.1641093</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Marine Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A novel reinforcement learning framework-based path planning algorithm for unmanned surface vehicle</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Mou</surname>
<given-names>Jianhui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shi</surname>
<given-names>Bo</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3080970/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Bo</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Chengcheng</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Yangwei</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhong</surname>
<given-names>Fusheng</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zheng</surname>
<given-names>Li</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Jian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Junjie</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>School of Electromechanical and Automotive Engineering, Yantai University</institution>, <addr-line>Yantai</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>National CAD Supported Software Engineering Centre, School of Mechanical Science and Engineering, Huazhong University of Science and Technology</institution>, <addr-line>Wuhan</addr-line>,&#xa0;<country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Suzhou Tongyuan Software &amp; Control Technology Co., Ltd.</institution>, <addr-line>Suzhou</addr-line>,&#xa0;<country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Chengbo Wang, University of Science and Technology of China, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Miao Gao, Tianjin University, China</p>
<p>Jiao Liu, Ningbo University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Junjie Li, <email xlink:href="mailto:lijunjie@ytu.edu.cn">lijunjie@ytu.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>01</day>
<month>08</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1641093</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Mou, Shi, Wang, Yu, Wang, Zhong, Zheng, Wang and Li.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Mou, Shi, Wang, Yu, Wang, Zhong, Zheng, Wang and Li</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Unmanned surface vehicles (USVs) nowadays have been widely used in ocean observation missions, helping researchers to monitor climate change, collect environmental data, and observe marine ecosystem processes. However, path planning for USVs often faces several inherent difficulties during ocean observation missions: high dependence on environmental information, long convergence time, and low-quality generated paths. To solve these problems, this article proposes a novel artificial potential field-heuristic reward-averaging deep Q-network (APF-RADQN) framework-based path planning algorithm, aiming at finding optimal paths for USVs. First, the USV path planning is modeled as a Markov decision process (MDP). Second, a comprehensive reward function incorporating artificial potential field (APF) inspiration is designed to guide the USV to reach the target region. Subsequently, an optimized deep neural network with a reward-averaging strategy is constructed to effectively enhance the learning and convergence speed of the algorithm, thus further improving the global search capability and interface performance of USV path planning. In addition, the Bezier curve is applied to make the generated path more feasible. Finally, the effectiveness of the proposed algorithm is verified by comparing it with the DQN, A*, and APF algorithms in simulation experiments. Simulation results demonstrate that the APF-RADQN improves the interface ability and path quality, significantly enhancing the USV navigation safety and ocean observation mission operation efficiency.</p>
</abstract>
<kwd-group>
<kwd>unmanned surface vehicles</kwd>
<kwd>reinforcement learning</kwd>
<kwd>deep Q-learning algorithm</kwd>
<kwd>artificial potential field algorithm</kwd>
<kwd>path planning</kwd>
</kwd-group>
<counts>
<fig-count count="11"/>
<table-count count="2"/>
<equation-count count="28"/>
<ref-count count="48"/>
<page-count count="16"/>
<word-count count="9101"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Ocean Observation</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>In recent years, ocean observation missions such as collecting high-resolution air-sea observations (<xref ref-type="bibr" rid="B37">Wills et&#xa0;al., 2023</xref>), oceanic data collection (<xref ref-type="bibr" rid="B46">Zhao and Bai, 2023</xref>), and marine ecosystem monitoring (<xref ref-type="bibr" rid="B8">Handegard et&#xa0;al., 2024</xref>) are deploying more and more unmanned surface vehicles (USVs) due to their excellent endurance, navigational stability, and maneuverability (<xref ref-type="bibr" rid="B47">Zhou et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B4">Chiodi et&#xa0;al., 2021</xref>). Nevertheless, during environmental observation missions, USVs sometimes need to sail in scenes where either obstacle is complexly distributed, or the environment lacks prior information. These situations may increase the time spent on missions and the probability of collisions, ultimately leading to increased resource consumption and, in some cases, mission failure. To ensure ocean observation missions are accomplished, it is necessary for USVs to get a feasible path generated by path planning algorithms. An effective path planning algorithm should not only generate collision-free routes to ensure safe navigation but also optimize mission duration. Therefore, it is crucial for USVs to adopt a suitable path planning algorithm in ocean observation missions.</p>
<p>There have been many studies on path planning algorithms for USVs, which can be broadly classified into the following three types: 1) traditional algorithms, 2) intelligent optimization algorithms, and 3) machine learning algorithms.</p>
<p>The A* algorithm (<xref ref-type="bibr" rid="B19">Ma et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B32">Wang et&#xa0;al., 2025a</xref>) and the artificial potential field (APF) algorithm (<xref ref-type="bibr" rid="B17">Liu et&#xa0;al., 2020</xref>) are widely used traditional path planning algorithms. Although the A* algorithm can efficiently identify feasible paths, the computation of the A* algorithm increases significantly as the exploration space increases. Thus, A* is unsuitable for open environments with complex obstacles (<xref ref-type="bibr" rid="B43">Yu et&#xa0;al., 2019</xref>). The APF algorithm refers to the concept of potential field in physics and utilizes the method of potential field descent to obtain the target route (<xref ref-type="bibr" rid="B16">Li et&#xa0;al., 2021</xref>). <xref ref-type="bibr" rid="B38">Wu et&#xa0;al. (2025)</xref> combined APF with RRT algorithm, proposing an artificial potential field Bidirectional-Rapidly Exploring Random Trees (APF B-RRT*) algorithm for multiple rolling path planning, which is used for collaborative path planning of multiple UAVs. However, with the increase of obstacles in the environment, the APF algorithm will become chaotic and complex and may even generate multiple zero potential energy points. These problems can easily make the path planning fall into local optimization, resulting in failure to reach the target point.</p>
<p>Commonly used intelligent optimization algorithms include particle swarm algorithms (PSO) (<xref ref-type="bibr" rid="B2">Antonakis et&#xa0;al., 2017</xref>) and their variants, genetic algorithms (GA) (<xref ref-type="bibr" rid="B29">Tsai et&#xa0;al., 2011</xref>) and their variants, and fuzzy logic algorithms, ant colony algorithms (<xref ref-type="bibr" rid="B24">Shi et&#xa0;al., 2022</xref>), etc. Although intelligent optimization algorithms require fewer computations and are easier to handle path planning under complex environmental constraints than traditional algorithms, they still have drawbacks, such as easily falling into local optimum and high dependency on environmental data. Therefore, traditional path planning methods and intelligent optimization path planning algorithms make it difficult to effectively solve path planning problems in complex and lack of prior knowledge environments.</p>
<p>Because of the high efficiency of data production and excellent ability to solve complex problems, machine learning algorithms have been utilized in the field of path planning in recent years. Machine learning based algorithms include deep learning (DL), reinforcement learning (RL), deep reinforcement learning (DRL), imitation learning methods, etc. <xref ref-type="bibr" rid="B1">Al-Kamil and Szabolcsi (2024)</xref> proposed a path planning algorithm that combined supervised and unsupervised learning methods. Literature (<xref ref-type="bibr" rid="B10">Ionescu, 2021</xref>; <xref ref-type="bibr" rid="B11">Jin et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B23">Santos et&#xa0;al., 2022</xref>), also used deep learning for solving path planning problems. Reinforcement learning (RL), as a kind of machine learning, allows USVs to learn the optimal driving strategy and obtain the optimal path in the continuous interaction with the environment. In terms of path planning, reinforcement learning methods show great potential for application in complex environments (<xref ref-type="bibr" rid="B22">Pflueger et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B31">Wang et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B33">Wang et&#xa0;al., 2025b</xref>). Literature (<xref ref-type="bibr" rid="B30">Wang et&#xa0;al., 2024c</xref>) used Q-learning and improved the greedy strategy in path planning and obstacle avoidance problems for robots. Literature (<xref ref-type="bibr" rid="B18">Low et&#xa0;al., 2019</xref>) proposed an algorithm combining Q-Learning and the Flower Pollination Algorithm (FPA), which was applied to Autonomous Mobile Robot (AMR) path planning, effectively addressing the slow convergence issue of traditional Q-Learning algorithms. However, the traditional Q-value table-based reinforcement learning method causes the Q-value table to increase rapidly when the state space and action space increase, leading to the occupation of a large amount of computational and storage resources, resulting in a decrease in computational speed and convergence speed (<xref ref-type="bibr" rid="B28">Tan et&#xa0;al., 2024</xref>). Therefore, signally using DL or RL still has problems that are highly dependency on historical data and poor capability of solving high dimensional state problems.</p>
<p>Deep reinforcement learning (DRL) combines reinforcement learning with deep learning. This combination makes DRL both having the decision-making ability of reinforcement learning and the generalized fitting ability of deep learning. Thus, the DRL algorithm solves the problem of dimensional explosion and enables reinforcement learning to deal with multi-dimensional state and action space. In recent years, deep reinforcement learning has been widely used in the autonomous control and path planning of unmanned vehicles such as UAVs, underwater robots, and other boats (<xref ref-type="bibr" rid="B42">Yang et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B5">Chu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B41">Xiaofei et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B44">Yu et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B35">Wang et&#xa0;al., 2024b</xref>). The deep Q-network (DQN) algorithm, as a kind of deep reinforcement learning, utilizes a deep neural network instead of a Q-value table. Since the DQN algorithm is simple and intuitive, many scholars have carried out research on path planning based on this algorithm and its variant algorithms. <xref ref-type="bibr" rid="B34">Wang et&#xa0;al. (2024a)</xref> proposed a risk and reliability critic-enhanced safe hierarchical reinforcement learning (RA-SHRL) algorithm based on the International Collision Avoidance Rules (COLREGs) for solving the collision avoidance decision in a multi-vessel encounter scenario. <xref ref-type="bibr" rid="B26">Su et&#xa0;al. (2022)</xref> proposed a target-oriented double deep Q-learning network (D2QN) based collision avoidance and trajectory planning algorithm for USVs to collect data from ocean detection networks. Yang et&#xa0;al. (<xref ref-type="bibr" rid="B41">Xiaofei et&#xa0;al., 2022</xref>) proposed a DQN-based global path planning algorithm for the amphibious USVs path planning problem. <xref ref-type="bibr" rid="B48">Zhu et&#xa0;al. (2021)</xref> proposed an Improved Dueling Deep Dual Q Network (IPD3QN) based on prioritized experience replay to solve the problem of slow and unstable convergence of traditional DQN algorithms for unmanned ships&#x2019; path planning. <xref ref-type="bibr" rid="B36">Wen et&#xa0;al. (2020)</xref> proposed a novel path planning method based on a neural network RL path system, which enables a mobile robot to navigate to the terminal area without colliding with any obstacles or robots, and the method has been successfully applied to the mobile robot experimental platform. It is worth noting that some researchers combine the APF algorithm with deep reinforcement learning algorithms to accelerate the training process. <xref ref-type="bibr" rid="B9">Hu et&#xa0;al. (2025)</xref> proposed a fuzzy A* quantum multi-stage Q-learning artificial potential field algorithm, which utilizes different algorithms in different path planning stages. <xref ref-type="bibr" rid="B15">Li et&#xa0;al. (2025)</xref> adopted the APF algorithm in the action selection policy. However, directly applying the DQN algorithm to USV path planning still has the following problems: 1) As the complexity of the environment rises, the learning efficiency and convergence speed of the DQN algorithm decreases. 2) The traditional DQN path planning method performs poorly in terms of safety, and the robot is at risk of getting too close to the edges of the obstacles. 3) Paths planned by traditional reinforcement learning usually have low smoothness.</p>
<p>In ocean observation missions, the high complexity of obstacles&#x2019; distribution complicates traditional path planning methods, making it difficult to guarantee both safety and efficiency. Therefore, an algorithm that can avoid collisions with obstacles while maintaining path efficiency is proposed. This study introduces a path planning algorithm based on artificial potential field-heuristic reward-averaging deep Q-network (APF-RADQN) framework, which incorporates collision avoidance awareness using a comprehensive reward function. After optimizing the path, the Bezier curve is applied to smooth the path, ensuring the smoothness and feasibility of the planned path. The main contribution points of this paper are summarized as follows:</p>
<list list-type="order">
<list-item>
<p>A novel comprehensive reward function is established, which incorporates the risk factors of multiple obstacles distributed environments, ensuring that the USV reaches the target position while avoiding obstacles in a complex environment.</p>
</list-item>
<list-item>
<p>The APF-RADQN framework is proposed, which incorporates a novel network update mechanism and path smoothing method to enhance both the convergence speed and path smoothness of the algorithm.</p>
</list-item>
<list-item>
<p>Experiments are designed for multiple scenarios, which show that the proposed APF-RADQN algorithm outperforms both the DQN algorithm and traditional algorithms listed in complex scenarios.</p>
</list-item>
</list>
<p>The remainder of the article is organized as follows. Section 2 describes the Markov decision process (MDP) of the USV path planning problem, the DQN algorithm, and the APF algorithm. Section 3 describes the application of the Markov model for USV path planning, the comprehensive reward function, and the APF-RADQN framework to USV path planning. Section 4 describes comparison experiments with several algorithms in different environments. Section 5 presents the conclusion and future works.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<p>USV path planning, as shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>, refers to the automatic generation of optimal routes by USVs from the starting point (SP) to the target point (TP) without colliding with obstacles such as reefs and buoys in a smooth manner based on environmental information and numerous constraints in the absence of human intervention (<xref ref-type="bibr" rid="B14">Lan et&#xa0;al., 2022</xref>). Through the related literature (<xref ref-type="bibr" rid="B40">Xi et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B45">Zhang et&#xa0;al., 2019</xref>), the USV path planning process can be classified as a serious MDP. Specifically, the USV takes the action only related to the current state. After adopting the chosen action, the USV obtains a reward form the environment and enters the next state. The USV expects to learn an optimal policy in the continuous interaction with the environment. The optimal policy maximizes the cumulative rewards to the USV and ultimately generates an optimal path based on this policy. Based on the above description, the MDP and the DQN algorithm are described below.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>USV path planning.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g001.tif">
<alt-text content-type="machine-generated">Diagram illustrating a ship's navigation from a start point to a target point, marked by a star. The ship avoids obstacles, shown as rocks, following an alternate green path. A red path with a danger sign indicates a hazardous route. A buoy is also depicted. Axes labeled x and y are included.</alt-text>
</graphic>
</fig>
<sec id="s2_1">
<label>2.1</label>
<title>Markov decision process</title>
<p>In an MDP, a USV can be abstracted into an agent with the corresponding attributes of the USV. MDP can be expressed as a four-tuple <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">[</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>A</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>,where <italic>S</italic> is the finite set of states in the environment, <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>is the state at the time <italic>t</italic>, <italic>A</italic> presents the set of all actions that the agent can execute, <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>presents the action performed by the agent at the time <italic>t</italic>, <italic>P</italic> presents the state transfer probability, averaging the probability that the state <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>moves to the next state <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <italic>R</italic> is the reward function. Based on the observed current state <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the agent chooses the action <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>to execute and enters the next state <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and the environment gives the agent a reward <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>when the agent enters the state <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> illustrates the MDP that models the interaction between the agent and the environment.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>USV Markov decision process.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g002.tif">
<alt-text content-type="machine-generated">Diagram illustrating the interaction between a USV (Unmanned Surface Vehicle) agent and its environment. The USV receives states \(S_t\) from the environment, takes actions \(A_t\), and receives rewards \(R_t\). The loop continues with future states \(S_{t+1}\) and rewards \(R_{t+1}\). The environment is depicted with a photo of a seascape.</alt-text>
</graphic>
</fig>
<p>The agent aims to acquire the optimal policy through the learning process. The optimal policy enables the agent to maximize the return for every episode. For each episode, the return <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>is calculated as the discounted sum of rewards <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The return <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be expressed as <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>&#x221e;</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;</mml:mtext>
<mml:mi>&#x3b3;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="false">[</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the reward that the agent obtains at step <inline-formula>
<mml:math display="inline" id="im15">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math display="inline" id="im16">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> is the discount factor, which represents the relationship between the immediate reward and the long-term reward.</p>
<p>The state value function is the value of state <italic>s</italic> in the environment under policy <inline-formula>
<mml:math display="inline" id="im17">
<mml:mi>&#x3c0;</mml:mi>
</mml:math>
</inline-formula>. State value function can be expressed as <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">[</mml:mo>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo stretchy="false">|</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">[</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>&#x221e;</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mo stretchy="false">|</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Similarly, action value function <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the value of taking action <italic>a</italic> in state <italic>s</italic> under policy <inline-formula>
<mml:math display="inline" id="im19">
<mml:mi>&#x3c0;</mml:mi>
</mml:math>
</inline-formula>. The action value function <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> presents the expected cumulative reward when USV takes action <italic>a</italic> in state <italic>s</italic> under policy <inline-formula>
<mml:math display="inline" id="im21">
<mml:mi>&#x3c0;</mml:mi>
</mml:math>
</inline-formula>. The action value function <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> can be expressed as <xref ref-type="disp-formula" rid="eq3">Equation 3</xref>:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">[</mml:mo>
<mml:msub>
<mml:mi>G</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo stretchy="false">|</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>E</mml:mi>
<mml:mi>&#x3c0;</mml:mi>
</mml:msub>
<mml:mo stretchy="false">[</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>&#x221e;</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mo stretchy="false">|</mml:mo>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The Bellman optimality equation for the state value function and the action value function can be expressed as <xref ref-type="disp-formula" rid="eq4">Equations 4</xref>, <xref ref-type="disp-formula" rid="eq5">5</xref>:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>*</mml:mo>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mi>a</mml:mi>
</mml:munder>
<mml:mi>E</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>*</mml:mo>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>*</mml:mo>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mi>a</mml:mi>
</mml:munder>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>*</mml:mo>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mo>*</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>*</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> present the maximum state value and action value of the state <italic>s</italic> under the optimal policy <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>*</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The optimal policy can be determined by <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>arg</mml:mi>
<mml:mi>max</mml:mi>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2217;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Deep Q-network</title>
<p>In this section, the DQN algorithm is used to solve the USV path planning problem. The DQN algorithm, as a representative deep reinforcement learning method, effectively combines Q-learning with deep learning. This combination enables the DQN algorithm to obtain the optimal policy in a high dimensional state space without relying on prior information about the environment (<xref ref-type="bibr" rid="B20">Mnih et&#xa0;al., 2015</xref>). Therefore, the DQN algorithm can enable agents to generate optimal paths under conditions of limited external environment information.</p>
<p>The neural network is used to fit the Q-table in the DQN algorithm, as shown in <xref ref-type="disp-formula" rid="eq6">Equation 6</xref>. During the training process, the agent constantly interacts with the environment to generate experience data. Specifically, the agent uses the Q-evaluation network to get the next action <italic>a</italic> at the state <italic>s</italic>, and executes the action according to a certain action selection probability <inline-formula>
<mml:math display="inline" id="im27">
<mml:mi>&#x3f5;</mml:mi>
</mml:math>
</inline-formula>. The commonly used action selection probability function is <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mtext>-greedy</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>,which can be expressed as <xref ref-type="disp-formula" rid="eq6">Equation 6</xref>:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>|</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtext>for&#xa0;the&#xa0;greedy&#xa0;action</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mi>&#x3b5;</mml:mi>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtext>for&#xa0;the&#xa0;other&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mtext>&#xa0;actions</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im29">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>|</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the policy that agent will take at the state <italic>s</italic>, and <inline-formula>
<mml:math display="inline" id="im30">
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the number of actions associated with <italic>s</italic>
</p>
<p>After executing the action, the agent enters into the next state <inline-formula>
<mml:math display="inline" id="im31">
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula> while obtaining a reward <italic>r</italic>. Through this process, the experience data of the interaction between the agent and the environment are stored in the form of tuples <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Then, the experience caches in the replay buffer.</p>
<p>The difference between the Q-evaluation network and the Q-target network is the way to use empirical data and update the network. Specifically, the Q-evaluation network directly derives state values based on the current state <italic>s</italic>, while the Q-target network calculates state values based on the next state <inline-formula>
<mml:math display="inline" id="im33">
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula> and the reward <italic>r</italic> earned. By calculating the error between the Q-evaluation and Q-target networks, the parameters of the Q-evaluation network can be updated. While for the parameters in the Q-target network, the network parameters of the Q-evaluation are directly copied to the Q-target network at a certain time interval. The calculation of the Q-target network can be expressed by the <xref ref-type="disp-formula" rid="eq7">Equation 7</xref>:</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mi>a</mml:mi>
</mml:munder>
<mml:mi>Q</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:munder>
<mml:mi>Q</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the maximum value of the Q-evaluation network output, <inline-formula>
<mml:math display="inline" id="im35">
<mml:msup>
<mml:mi>a</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula> is the action taken by the agent at the state <inline-formula>
<mml:math display="inline" id="im36">
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula>, and <inline-formula>
<mml:math display="inline" id="im37">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the Q-target network&#x2019;s parameter.</p>
<p>The loss function <inline-formula>
<mml:math display="inline" id="im38">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is defined by the mean square deviation between two action values predicted by Q-evaluation network and Q-target network, which can be expressed as <xref ref-type="disp-formula" rid="eq8">Equation 8</xref>:</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo stretchy="false">[</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mtext>eval</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im39">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mtext>eval</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the output of the Q-evaluation network and <inline-formula>
<mml:math display="inline" id="im40">
<mml:mi>&#x3b8;</mml:mi>
</mml:math>
</inline-formula> denotes the parameters of the network.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Artificial potential field</title>
<p>The Artificial Potential Field (APF) algorithm first appeared in a paper published by Oussama Khatib in 1985 (<xref ref-type="bibr" rid="B12">Khatib, 1985</xref>). The APF algorithm assumes that there exists a gravitational force on the target point and a repulsive force on the obstacle point. The gravitation force can attract the agent to go to the target point, while repulsive force makes the agent stay away from obstacles and avoid colliding with obstacles. The agent is subjected to the field force at every point on the map. The field force is equal to the sum of the gravitational and repulsive forces at that point.</p>
<p>The gravitational field can be expressed as <xref ref-type="disp-formula" rid="eq9">Equation 9</xref>:</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mtext>att</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>&#x3be;</mml:mi>
<mml:msup>
<mml:mi>&#x3c1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mtext>goal</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im41">
<mml:mi>&#x3be;</mml:mi>
</mml:math>
</inline-formula> is the gravitational coefficient, <inline-formula>
<mml:math display="inline" id="im42">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mtext>goal</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the distance between the location <italic>h</italic> of the agent and the location <inline-formula>
<mml:math display="inline" id="im43">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mtext>goal</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the target point.</p>
<p>The gravitational force exerted by the target point on the agent can be derived by calculating the gradient of the gravitational field. The gravitational force can be expressed as <xref ref-type="disp-formula" rid="eq10">Equation 10</xref>:</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mtext>att</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3be;</mml:mi>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mtext>goal</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In order to avoid the collision between the agent and obstacles, the repulsive field exhibits a different property from the gravitational field. For the repulsive field, the closer the agent is to the obstacle, the larger the repulsive field becomes. The repulsive field can be expressed as <xref ref-type="disp-formula" rid="eq11">Equation 11</xref>:</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:msub>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
<mml:mi>k</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&lt;</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&gt;</mml:mo>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>k</italic> is the repulsion coefficient, <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is the distance between the location <italic>h</italic> of the agent and the location <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the <italic>i</italic>th obstacle <inline-formula>
<mml:math display="inline" id="im46">
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the maximum repulsive force range of the obstacle <inline-formula>
<mml:math display="inline" id="im48">
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. When the agent leaves the repulsive force range of the obstacle, the agent is no longer subject to the repulsive force of this obstacle.</p>
<p>The repulsive force generated by obstacles can be derives by calculating the gradient of the repulsive force field. The repulsive force can be expressed as <xref ref-type="disp-formula" rid="eq12">Equation 12</xref>:</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo stretchy="false">)</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2207;</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&lt;</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&#x2264;</mml:mo>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>&gt;</mml:mo>
<mml:msub>
<mml:mi>&#x3c1;</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The combined force acting on the agent is a superposition of gravitational and repulsive forces, so the combined force acting on the agent can be written as <xref ref-type="disp-formula" rid="eq13">Equation 13</xref>:</p>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mtext>att</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mtext>res</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The gravitational force generated by the target point will make the agent move toward the target point, and the repulsive force generated by the obstacle will prevent the agent from approaching the obstacle. Through the superposition of gravitational force and repulsive force, the agent can avoid the obstacle and reach the target location.</p>
<p>However, as shown in <xref ref-type="disp-formula" rid="eq13">Equation 13</xref>, the agent is ultimately subject to the combined force of the gravitational force and the repulsive force. In some special cases, there may be a situation where the repulsive force is balanced with the gravitational force, as shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>. In addition, when obstacles are gathered at the annex of the target point, it is possible that the combined repulsive force of the obstacles will be greater than the gravitational force at the target location, and thus, the target location will no longer be the point with the smallest potential field in the whole map, resulting in the agent not being able to reach the target location.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Gravitational force and repulsive force balance.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g003.tif">
<alt-text content-type="machine-generated">Graph showing a boat navigating between two circular obstacles towards a target marked by a star. Forces are represented by arrows: \( F_{\text{res1}} \), \( F_{\text{res2}} \), and \( F_{\text{att}} \). The x and y axes are labeled, indicating directions.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>APF-RADQN-based path planning algorithm</title>
<p>In this section, a MDP model of the USV is first established. The MDP model primarily consists of the state space, the action space, and the reward function. Secondly, an APF-RADQN framework based on improved network error calculation is proposed, which is used to solve the MDP model of USV path planning. Then, the optimal path is obtained by the path smoothing method.</p>
<sec id="s3_1">
<label>3.1</label>
<title>MDP model for USV route planning</title>
<p>1. State space: Grid map-based metric modeling is the mainstream environmental map building method (<xref ref-type="bibr" rid="B13">Koval et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B39">Wu et&#xa0;al., 2022</xref>). Meanwhile, unlike UAVs that can float or dive, USVs only move on the surface of the water. In that case, USVs are mainly concerned with the environmental state information of the plane of movement where they are located. Therefore, the three-dimensional spatial environment can be converted into a two-dimensional planar environment, as shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. The navigation environment of the USV is set as a two-dimensional grid map environment of 60m*60m. According to the distribution of obstacles in the environment, the obstacles are represented by black grids, and the area that can be freely passed is represented by white grids. Each of the grid is a square with a width of 1 meter. In particular, in order to facilitate the construction of the simulation map environment and to ensure the safe navigation of the USV, obstacles that do not occupy a complete grid in actuality are regarded as occupying a complete grid in the gridded map. Additionally, the grid map-based environment is determined in this study, which meaning when the agent takes the action <italic>a</italic> at state <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, then the next state <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the agent is determined. Thus, the state transfer probability <italic>P</italic> can be expressed as <inline-formula>
<mml:math display="inline" id="im51">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>|</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Conversion of a 3D spatial environment into a 2D planar environment.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g004.tif">
<alt-text content-type="machine-generated">3D map on the left shows a pyramid shape on a grid with an &#x201c;SP&#x201d; start point marked by an orange circle and a &#x201c;TP&#x201d; target point marked by a red star. Arrows indicate movement from SP to TP. On the right, a 2D map translates the 3D structure into a top-down grid, with SP and TP similarly marked, demonstrating the same movement path.</alt-text>
</graphic>
</fig>
<p>The Euclidean distance is used to measure the distance between the USV and the target point when calculating the path length. So, the coordinate position <inline-formula>
<mml:math display="inline" id="im52">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> of the USV and the distance to the target point <inline-formula>
<mml:math display="inline" id="im53">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are introduced into the state space. The state space equation of the USV can be expressed as <xref ref-type="disp-formula" rid="eq14">Equation 14</xref>:</p>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>2. Action space: typically, the propulsion, steering and other motion states of a USV are realized by coordinating the thrust of propellers with the deflection angle of its rudder. Although the navigation process of USVs in real environments is continuous, the navigation of USVs within a period of time can be regarded as movement within a grid. Therefore, the navigation process of USVs in real environments can be converted into a series of grid movements of agents in grid maps. Moreover, in order to simplify the USV&#x2019;s motion model, in the path planning process, it is assumed that the speed of the USV is certain, and at the same time, the USV&#x2019;s action space is decomposed into 8 directions <inline-formula>
<mml:math display="inline" id="im54">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>=</mml:mo>
<mml:mo>{</mml:mo>
<mml:mtext>right,&#xa0;down,&#xa0;left,&#xa0;up,&#xa0;lower&#xa0;right</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;upper&#xa0;right,&#xa0;lower&#xa0;left</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;upper&#xa0;left</mml:mtext>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., the action space. The specific action space is shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>. Based on the above assumptions, the movement rules of USV in the gridded map environment can be represented as <xref ref-type="disp-formula" rid="eq15">Equation 15</xref>:</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>USV action space.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g005.tif">
<alt-text content-type="machine-generated">Diagram showing a grid with a boat in the center and labeled arrows pointing in eight directions: up, down, left, right, upper left, upper right, lower left, and lower right. Axes labeled &#x201c;x&#x201d; and &#x201c;y&#x201d;.</alt-text>
</graphic>
</fig>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im55">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the current position state of the USV, <inline-formula>
<mml:math display="inline" id="im56">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the next position state of the USV, and <inline-formula>
<mml:math display="inline" id="im57">
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the width of the unit grid. It is important to emphasize that each action performed by the USV in the gridded map only allows it to move to an adjacent grid.</p>
<p>3. Reward Function: In the MDP framework, the reward function is used to evaluate the value of an action <italic>a</italic> taken by the agent. In previous studies, an agent receives a positive reward when it reaches the target location and a negative reward for entering a forbidden zone such as an obstacle or a boundary. However, under this reward distribution, all other states receive zero reward, which causes the problem of reward sparsity in reinforcement learning. This problem leads to difficulties for reinforcement learning algorithms to find feasible solutions. Therefore, this paper proposes a composite reward function <inline-formula>
<mml:math display="inline" id="im58">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The proposed reward function makes it possible to receive a reward for agent at all state during the training process. The composite reward function can be expressed as <xref ref-type="disp-formula" rid="eq16">Equation 16</xref>:</p>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im59">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the target reward function, <inline-formula>
<mml:math display="inline" id="im60">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the obstacle reward function, <inline-formula>
<mml:math display="inline" id="im61">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the boundary reward function, and <inline-formula>
<mml:math display="inline" id="im62">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the potential field reward function.</p>
<p>Specifically, <inline-formula>
<mml:math display="inline" id="im63">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is usually set as a positive reward to motivate the agent to reach the target position, and the target reward function can be calculated by the <xref ref-type="disp-formula" rid="eq17">Equation 17</xref>:</p>
<disp-formula id="eq17">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msqrt>
<mml:mn>2</mml:mn>
</mml:msqrt>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mtext>grid</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtext>otherwise</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
<mml:mtext>&#xa0;&#xa0;&#xa0;&#xa0;</mml:mtext>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im64">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a positive reward value, which is available when the agent enters the target position. To improve the training efficiency of the model, the task is considered completed when the agent reaches a certain range of the target point. For the grid map, the task is considered completed when the agent enters the grid where the target point is located.</p>
<p>where DT is the Euclidean distance from the agent to the target point, which can be expressed by <xref ref-type="disp-formula" rid="eq18">Equation 18</xref>:</p>
<disp-formula id="eq18">
<label>(18)</label>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:math>
</disp-formula>
<p>During the navigation process, agent should avoid the collision with obstacles in the environment. In grid-based map, the agent should not drive into the grids labeled as obstacles on the map. The obstacle reward function <inline-formula>
<mml:math display="inline" id="im65">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can effectively avoid the agent from colliding with obstacles in the grided-based movement, and its expression as <xref ref-type="disp-formula" rid="eq19">Equation 19</xref>:</p>
<disp-formula id="eq19">
<label>(19)</label>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtext>OD</mml:mtext>
<mml:mo>&lt;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtext>OD</mml:mtext>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im66">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a negative value indicating the penalty for the agent to enter the range of an obstacle, OD is the Euclidean distance from the agent to the closest obstacle to the agent, and <inline-formula>
<mml:math display="inline" id="im67">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the safe distance that the agent should maintain from the obstacle.</p>
<p>Boundary rewards are used to limit the area of activity of an agent. Ensure that the agent operates within a predefined range of the environment. Boundary rewards can be defined as <xref ref-type="disp-formula" rid="eq20">Equation 20</xref>:</p>
<disp-formula id="eq20">
<label>(20)</label>
<mml:math display="block" id="M20">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&gt;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mtext>)&#xa0;or&#xa0;</mml:mtext>
<mml:mi>x</mml:mi>
<mml:mo>&lt;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&gt;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mtext>)&#xa0;or&#xa0;</mml:mtext>
<mml:mi>y</mml:mi>
<mml:mo>&lt;</mml:mo>
<mml:mi>min</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mtext>otherwise</mml:mtext>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where reward <inline-formula>
<mml:math display="inline" id="im68">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a negative value, <inline-formula>
<mml:math display="inline" id="im69">
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im70">
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> are the maximum and minimum horizontal coordinates of the environmental map, <inline-formula>
<mml:math display="inline" id="im71">
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im72">
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> are the maximum and minimum vertical coordinates of the environmental map.</p>
<p>When the agent explores the environment, if rewards are only set at key points that directly affect the success or failure of the task, such as target points, boundaries and obstacles, while rewards are zero at other relatively unimportant locations, it will lead to slower convergence of the algorithm, or even ultimately fail to obtain a valid path. Therefore, continuous rewards are needed to provide guidance to the agent during its exploration of the environment. The successive rewards should have the ability to guide the agent to the target location while avoiding factors such as obstacles and boundaries. As shown in <xref ref-type="disp-formula" rid="eq10">Equations 10</xref>, <xref ref-type="disp-formula" rid="eq12">12</xref> and <xref ref-type="disp-formula" rid="eq13">13</xref>, the APF method is characterized by a continuous distribution of potential field forces in the map environment. Specifically, potential field forces point to the target point at locations away from obstacles, and point in the direction away from obstacles near obstacles. The motion direction of the agent in the environment is shown in <xref ref-type="disp-formula" rid="eq15">Equation 15</xref>. According to the motion direction of the agent, the motion of the agent can be rewritten in the form of vector as <xref ref-type="disp-formula" rid="eq21">Equation 21</xref>:</p>
<disp-formula id="eq21">
<label>(21)</label>
<mml:math display="block" id="M21">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>right</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>down</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>left</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>up</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>lower_right</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>upper_right</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>lower_left</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>a</mml:mi>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mtext>upper_left</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>grid</mml:mtext>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Therefore, in this paper, a potential field reward function is proposed in combination with the APF method, the reward function is expressed as <xref ref-type="disp-formula" rid="eq22">Equation 22</xref>:</p>
<disp-formula id="eq22">
<label>(22)</label>
<mml:math display="block" id="M22">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
<mml:mo>&#xb7;</mml:mo>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mi>cos</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where the reward <inline-formula>
<mml:math display="inline" id="im73">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the dot product of the vector <inline-formula>
<mml:math display="inline" id="im74">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and vector <inline-formula>
<mml:math display="inline" id="im75">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, the vector <inline-formula>
<mml:math display="inline" id="im76">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is the action taken by the agent in the state <inline-formula>
<mml:math display="inline" id="im77">
<mml:mi>s</mml:mi>
</mml:math>
</inline-formula>, vector <inline-formula>
<mml:math display="inline" id="im78">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2192;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> is the potential force on the agent in the state <inline-formula>
<mml:math display="inline" id="im79">
<mml:mi>s</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im80">
<mml:mrow>
<mml:mi>cos</mml:mi>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the cosine of the angle between two vectors, when <inline-formula>
<mml:math display="inline" id="im81">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>&lt;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&lt;</mml:mo>
<mml:mfrac>
<mml:mi>&#x3c0;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, the reward <inline-formula>
<mml:math display="inline" id="im82">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is positive, when <inline-formula>
<mml:math display="inline" id="im83">
<mml:mrow>
<mml:mfrac>
<mml:mi>&#x3c0;</mml:mi>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>&lt;</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>&lt;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the reward is negative. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> shows the reward <inline-formula>
<mml:math display="inline" id="im84">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> distribution. In the situation of the <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>, supposing the APF force is the unit, the angle <inline-formula>
<mml:math display="inline" id="im85">
<mml:mi>&#x3b8;</mml:mi>
</mml:math>
</inline-formula> is <inline-formula>
<mml:math display="inline" id="im86">
<mml:mrow>
<mml:mfrac>
<mml:mi>&#x3c0;</mml:mi>
<mml:mn>6</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, and the action vectors are defined by the <xref ref-type="disp-formula" rid="eq21">Equation 21</xref>.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Distribution of rewards.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g006.tif">
<alt-text content-type="machine-generated">Grid with eight quadrants showing reward values, varying from negative to positive, represented by colors from red to blue. A boat illustration in the center is surrounded by yellow arrows indicating potential directions. The x and y axes are marked.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="disp-formula" rid="eq22">Equation 22</xref> demonstrates the relationship between action and reward: the reward increases when the agent&#x2019;s action direction aligns with the artificial potential field; The reward decreases as the action deviates from the potential field&#x2019;s guidance; the reward transitions to negative values if the agent&#x2019;s motion opposes the potential field direction. Through the guidance of the artificial potential field, the agent can obtain a feasible path more quickly.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>APF-RADQN framework for USVs</title>
<p>1. Deep Neural Networks: The DQN algorithm uses deep neural networks (DNN) to approximate the action value function (<xref ref-type="bibr" rid="B7">Hadi et&#xa0;al., 2022</xref>). In addition, both the Q-evaluation network and the Q-target network use the same DNN network structure. The DNN structure consists of an input layer, several hidden layers and an output layer. The network parameters are trained through supervised learning. The input to the neural network consists of two parts: 1) the environmental information collected by the sensors and 2) the positional information of the USV. These inputs are processed through the fully connected layer and finally the computed results are output to the output layer. The output layer gives the Q-values of all executable actions and finally selects the action with the largest Q-value as the output of the network. In this network structure, the rectified liner unit (ReLu) function is used as an activation function between the input layer and each hidden layer.</p>
<p>The ReLu function can be expressed as <xref ref-type="disp-formula" rid="eq23">Equation 23</xref>:</p>
<disp-formula id="eq23">
<label>(23)</label>
<mml:math display="block" id="M23">
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3be;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>&gt;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>To improve the learning efficiency of the DNN, the adaptive moment estimation (Adam) algorithm is used to optimize the learning rate. The Adam optimizer enables adaptive updating of the learning rate (<xref ref-type="bibr" rid="B3">Chen et&#xa0;al., 2024</xref>). Meanwhile, Adam optimizer introduces a dynamic term to improve the stochastic gradient descent of the neural network. Subsequently, the current network is optimized through integration with the gradient explosion prevention method, thereby enhancing the predictive capability of the DQN algorithm.</p>
<p>2. Network reward-averaging strategy: as a key factor in reinforcement learning, rewards influence the learning efficiency and eventual convergence of the algorithm (<xref ref-type="bibr" rid="B6">De Moor et&#xa0;al., 2022</xref>). Agents continuously acquire rewards in their interactions with the environment and store these rewards in the experience replay buffer. During the network training process, these reward-contained experiences are also continuously used as training inputs in the process of optimizing the deep network. Recent studies show that optimizing the reward calculation in the DRL training process can speed up learning and improve the convergence of the network. (<xref ref-type="bibr" rid="B21">Naik et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B27">Sutton and Barto, 2020</xref>). To reduce the variance of the reward signal, the reward in each training step subtracts from the average reward. In the DQN algorithm, the network is updated by computing the temporal difference (TD) error between the Q-evaluation and Q-target values. Q-target value is calculated by reward and discounted action value. Therefore, when calculating the Q-target value and TD error, the reward obtained from the reward-averaging strategy can replace the original reward received from the experience. The Q-target network and the loss function, <xref ref-type="disp-formula" rid="eq7">Equations 7</xref>, <xref ref-type="disp-formula" rid="eq8">8</xref> can be rewritten as <xref ref-type="disp-formula" rid="eq24">Equations 24</xref>, <xref ref-type="disp-formula" rid="eq25">25</xref>:</p>
<disp-formula id="eq24">
<label>(24)</label>
<mml:math display="block" id="M24">
<mml:mrow>
<mml:msub>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mi>a</mml:mi>
</mml:munder>
<mml:mi>Q</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>;</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq25">
<label>(25)</label>
<mml:math display="block" id="M25">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mtext>eval</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im87">
<mml:mrow>
<mml:msub>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the output of the target network subtracting the mean value of the rewards, and <inline-formula>
<mml:math display="inline" id="im88">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the mean value of the historical rewards, which can be expressed by the <xref ref-type="disp-formula" rid="eq26">Equation 26</xref>:</p>
<disp-formula id="eq26">
<label>(26)</label>
<mml:math display="block" id="M26">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>r</mml:mi>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b7;</mml:mi>
<mml:mtext>mean</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im89">
<mml:mi>&#x3b7;</mml:mi>
</mml:math>
</inline-formula> is the learning coefficient of the reward mean, <inline-formula>
<mml:math display="inline" id="im90">
<mml:mi>&#x3b4;</mml:mi>
</mml:math>
</inline-formula> is the difference between the <inline-formula>
<mml:math display="inline" id="im91">
<mml:mrow>
<mml:msub>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mtext>target</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> value and the <inline-formula>
<mml:math display="inline" id="im92">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mtext>eval</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> value, <inline-formula>
<mml:math display="inline" id="im93">
<mml:mi>r</mml:mi>
</mml:math>
</inline-formula> is the total reward corresponding to the training step.</p>
<p>3. Path smoothing: The output of the APF-RADQN algorithm is discrete, while the navigation of the USV is continuous. The points generated by the algorithm need to be smoothed to make the paths closer to the actual USV navigation. Bezier curve has the advantages of high controllability, good path smoothing, and less occupied computation (<xref ref-type="bibr" rid="B25">Song et&#xa0;al., 2021</xref>). Bezier curve is widely used in curve smoothing related graphical processing work. Therefore, the introduction of Bezier curve into the framework of APF-RADQN algorithm can realize the end-to-end smoothing path planning. the computational expressions of Bezier curve from the first order to the third order are as <xref ref-type="disp-formula" rid="eq27">Equation 27</xref>:</p>
<disp-formula id="eq27">
<label>(27)</label>
<mml:math display="block" id="M27">
<mml:mrow>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>0</mml:mn>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mi>t</mml:mi>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im94">
<mml:mrow>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>0</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents the <italic>i</italic>th initial control point, <inline-formula>
<mml:math display="inline" id="im95">
<mml:mrow>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents the <italic>i</italic>th <italic>K</italic>-order control point (k=1, 2, &#x2026;, n), and <inline-formula>
<mml:math display="inline" id="im96">
<mml:mi>t</mml:mi>
</mml:math>
</inline-formula> represents the scale factor.</p>
<p>According to <xref ref-type="disp-formula" rid="eq27">Equation 27</xref>, the formula for <italic>n</italic>-order control points always contains two <italic>n</italic>-1 order control points, thus the formula for <italic>n</italic>-order control points can be finally calculated from the initial control points. Second-order Bezier curves ensure smoothness of paths and have faster computation speeds than higher-order Bezier curves. Therefore, multiple second-order Bezier curves are used in series for smoothing the path as <xref ref-type="disp-formula" rid="eq28">Equation 28</xref>:</p>
<disp-formula id="eq28">
<label>(28)</label>
<mml:math display="block" id="M28">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>0</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo stretchy="false">(</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>0</mml:mn>
</mml:msubsup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mn>0</mml:mn>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mo stretchy="false">[</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im97">
<mml:mrow>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mn>0</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im98">
<mml:mrow>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mn>0</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im99">
<mml:mrow>
<mml:msubsup>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mn>0</mml:mn>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> denote three consecutive initialization control points.</p>
<p>4. USV path planning process: in summary, the flowchart of the APF-RADQN algorithm applied to USV path planning is shown in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. The pseudocode of the algorithm is shown in <xref ref-type="statement" rid="algo1">
<bold>Algorithm 1</bold>
</xref> . The flow of Code 1 is depicted roughly as follows.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Algorithm flow chart.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g007.tif">
<alt-text content-type="machine-generated">Diagram illustrating a reinforcement learning process for an agent (USV) interacting with its environment. The environment includes a target point, obstacles, and a reward function with negative and positive rewards. The process involves a replay buffer, mini-batches, Q-evaluation, Q-target networks, updates of parameters (&#x3b8;), and a loss function. Arrows indicate the flow of states, actions, and updates.</alt-text>
</graphic>
</fig>
<p>Step 1: First, set the training parameters, mainly including state space, action space, reward, initial learning rate, etc. (line 1); second, establish Q-evaluation neural network and Q-target neural network; third, randomly initialize the parameters <inline-formula>
<mml:math display="inline" id="im100">
<mml:mi>&#x3b8;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im101">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, and make <inline-formula>
<mml:math display="inline" id="im102">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> (line 2).</p>
<p>Step 2: USV explores the environment. The environment is first reset before the start of each episode (line 5). During the exploration process, the current state inputs to the Q-evaluation network and the response action is calculated. The resulting action is executed according to the <inline-formula>
<mml:math display="inline" id="im103">
<mml:mrow>
<mml:mtext>&#x3f5;-greedy</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> policy. The USV interacts with the environment to receive rewards <italic>r</italic> and the next state <italic>s</italic> (lines 7-10).</p>
<p>Step 3: Update process of Q-evaluation and Q-target network. First, the sampled data collected during exploration are saved in the experience replay buffer, and then a batch of experience data are randomly selected from the experience replay buffer for updating the current network (lines 11-15). The Q-target network and the Q-evaluation network are computed using empirical data, and then the error between the networks is computed and updated <italic>rm</italic>. Second, the loss function is computed and the parameters of the Q-evaluation network are updated based on the backpropagation of the gradient of the neural network. Third, update the parameters of the Q-target network (lines 17-18).</p>
<p>Step 4: After multiple rounds of training, save the final training parameters of the network and the USV output optimal path (lines 21-23).</p>
<statement id="algo1">
<label>Algorithm 1</label>
<title>APF-RADQN Algorithm.</title>
<p>
<preformat>
1: <bold>Input</bold>: environment map; action set A; learning rate <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im104">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula></named-content>; discount factor <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im105">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula></named-content>; reward-averaging factor <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im106">
<mml:mi>&#x3b7;</mml:mi>
</mml:math>
</inline-formula></named-content>; experience replay pool maximum size D; mini-batch <italic>M</italic>; APF parameters <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im107">
<mml:mrow>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula></named-content>; exploration probability <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im108">
<mml:mi>&#x3f5;</mml:mi>
</mml:math>
</inline-formula></named-content>.&#xD;
2: Initialize network parameters <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im109">
<mml:mi>&#x3b8;</mml:mi>
</mml:math>
</inline-formula></named-content> and <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im110">
<mml:mrow>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula></named-content>, let <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im111">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula></named-content>, initialize policy <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im112">
<mml:mi>&#x3c0;</mml:mi>
</mml:math>
</inline-formula></named-content>, initialize <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im113">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula></named-content>, and initialize the replay buffer.&#xD;
3: Set maximum episode <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im114">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula></named-content> and maximum step <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im115">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</named-content>
&#xD;4: <bold>for</bold> episode = 1: <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im116">
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula></named-content>:
&#xD;5: Reset the environment.
&#xD;6: &#x2003;&#x2003;<bold>for</bold> step = 1: <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im117">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula></named-content>:
&#xD;7: &#x2003;&#x2003;&#x2003;Get the state <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im118">
<mml:mi>s</mml:mi>
</mml:math>
</inline-formula></named-content> from environment.
&#xD;8: &#x2003;&#x2003;&#x2003;Calculate the action value through the <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im119">
<mml:mrow>
<mml:msub>
<mml:mi>Q</mml:mi>
<mml:mrow>
<mml:mtext>eval</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula></named-content> network.
&#xD;9: &#x2003;&#x2003;&#x2003;Choose action according to <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im120">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mtext>-greedy</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula></named-content> algorithm.
&#xD;10: &#x2003;&#x2003;&#x2003;Receive reward <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im121">
<mml:mrow>
<mml:msub>
<mml:mi>R</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula></named-content> and go to <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im122">
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula></named-content>.
&#xD;11: &#x2003;&#x2003;&#x2003;Store transition <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im123">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>s</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula></named-content> into replay buffer <italic>d</italic>.
&#xD;12: &#x2003;&#x2003;&#x2003;<bold>if</bold> <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im124">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>&gt;</mml:mo>
<mml:mi>D</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula></named-content>:
&#xD;13: &#x2003;&#x2003;&#x2003;&#x2003;Remove the first transition from replay buffer.
&#xD;14: &#x2003;&#x2003;&#x2003;<bold>end if</bold>
&#xD;15: &#x2003;&#x2003;&#x2003;Sample a mini-batch transition <italic>M</italic> from replay buffer.
&#xD;16: &#x2003;&#x2003;&#x2003;Calculate the target <italic>Q</italic> value by (24).
&#xD;17: &#x2003;&#x2003;&#x2003;Calculate the loss function by (25) and reward mean by (26).
&#xD;18: &#x2003;&#x2003;&#x2003;Update the target network occasionally.&#xD;
19: &#x2003;&#x2003;<bold>end for</bold>
&#xD;20: <bold>end for</bold>
&#xD;21: Obtain policy <named-content content-type="inline-equation"><inline-formula>
<mml:math display="inline" id="im125">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>=</mml:mo>
<mml:mi>arg</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>Q</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msup>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula></named-content>.
&#xD;22: Smooth the path of USV by (28)
&#xD;23: <bold>Output</bold>: The trajectory of the USV
</preformat>
</p>
</statement>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experimental simulation</title>
<p>In order to verify the feasibility of the proposed APF-RADQN algorithm, the simulation model of the proposed APF-RADQN algorithm and the other three path planning algorithms is established in PyCharm platform, and three 60*60 dimension grid-based map environments are established for the visualization of the results. The hyperparameter configurations of the APF-RADQN algorithm are summarized in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The operating system used for testing algorithms is Windows 11. The hardware configuration: CPU is Intel i9 13900hx, GPU is Nvidia RTX 4060, 8G video memory. The proposed APF-RADQN algorithm is compared with the baseline algorithm in the same three unknown obstacle environments to verify the effectiveness and superiority of the APF-RADQN algorithm.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Algorithm parameters.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Name of parameters</th>
<th valign="middle" align="left">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Action space size</td>
<td valign="middle" align="left">8</td>
</tr>
<tr>
<td valign="middle" align="left">Learning rate <inline-formula>
<mml:math display="inline" id="im126">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula>
</td>
<td valign="middle" align="left">0.001</td>
</tr>
<tr>
<td valign="middle" align="left">Discount factor <inline-formula>
<mml:math display="inline" id="im127">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula>
</td>
<td valign="middle" align="left">0.96</td>
</tr>
<tr>
<td valign="middle" align="left">
<inline-formula>
<mml:math display="inline" id="im128">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mtext>-greedy</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula> value <inline-formula>
<mml:math display="inline" id="im129">
<mml:mi>&#x3f5;</mml:mi>
</mml:math>
</inline-formula>
</td>
<td valign="middle" align="left">0.01</td>
</tr>
<tr>
<td valign="middle" align="left">Replay buffer size</td>
<td valign="middle" align="left">30000</td>
</tr>
<tr>
<td valign="middle" align="left">Mini-batch size</td>
<td valign="middle" align="left">64</td>
</tr>
<tr>
<td valign="middle" align="left">Target network update step</td>
<td valign="middle" align="left">20</td>
</tr>
<tr>
<td valign="middle" align="left">Hidden layers size</td>
<td valign="middle" align="left">4</td>
</tr>
<tr>
<td valign="middle" align="left">Number of neurons</td>
<td valign="middle" align="left">256</td>
</tr>
<tr>
<td valign="middle" align="left">Activation function</td>
<td valign="middle" align="left">Relu</td>
</tr>
<tr>
<td valign="middle" align="left">APF parameters <inline-formula>
<mml:math display="inline" id="im130">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3be;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c1;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td valign="middle" align="left">(0.6, 0.05, 2)</td>
</tr>
<tr>
<td valign="middle" align="left">reward mean factor <inline-formula>
<mml:math display="inline" id="im131">
<mml:mi>&#x3b7;</mml:mi>
</mml:math>
</inline-formula>
</td>
<td valign="middle" align="left">0.001</td>
</tr>
<tr>
<td valign="middle" align="left">
<inline-formula>
<mml:math display="inline" id="im132">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td valign="middle" align="left">(10, -10, 10)</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s4_1">
<label>4.1</label>
<title>Baseline algorithms</title>
<p>Comparison algorithms are described as follows:</p>
<list list-type="order">
<list-item>
<p>APF: The APF algorithm quickly and flexibly plans an effective path by calculating the gradient of the potential field in the environment. The path trajectory can maintain a safe distance from obstacles.</p>
</list-item>
<list-item>
<p>A*: The A* algorithm accurately guides the robot to explore the environment by utilizing the principles of breadth-first search and best-first search. These search methods ensure the shortest path could be found in the limited state space.</p>
</list-item>
<list-item>
<p>DQN: The DQN algorithm obtains empirical data through the agent&#x2019;s continuous interaction with the environment, uses the empirical data to train the deep neural network, and finally obtains the optimal path through the neural network. Meanwhile, the state space, action space and reward function are the same of the proposed APF-RADQN algorithm.</p>
</list-item>
</list>
<p>Based on the advantages of APF algorithm, A* algorithm and DQN algorithm, the performance of different algorithms for planning path results in three aspects: 1) path length; 2) path smoothness; and 3) algorithmic inference time (IT) are evaluated to evaluate the superiority and effectiveness of the proposed APF-RADQN algorithm.</p>
<p>In addition, the computational complexity of all the compared algorithms is shown below. The computational complexity of the proposed APF-RADQN algorithm is <inline-formula>
<mml:math display="inline" id="im133">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>E</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>*</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, which is the same as that of the conventional DQN algorithm, where <italic>N</italic> is the step number in each episode and <italic>EP</italic> is the number of episode, The computational complexity of the A* algorithm is <inline-formula>
<mml:math display="inline" id="im134">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mi>log</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>,where <inline-formula>
<mml:math display="inline" id="im135">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of nodes. The computational complexity of the APF algorithm is <inline-formula>
<mml:math display="inline" id="im136">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>*</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im137">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the number of iterations and <italic>M</italic> denotes the number of obstacles.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Environmental simulation experiments and results</title>
<p>In this paper, three environments A, B, and C with different terrain features and obstacle densities are designed and tested. USV didn&#x2019;t learn three environments in advance to evaluate the performance of the proposed APF-RADQN algorithms and compare algorithms. First, three key performance indicators (KPIs) are established to evaluate the performance of various path planning algorithms. KPIs include average path length, average number of path corners, and algorithm inference time. The planned paths of these 4 algorithms in 3 different simulation environments are given in <xref ref-type="fig" rid="f8">
<bold>Figures&#xa0;8</bold>
</xref>&#x2013;<xref ref-type="fig" rid="f10">
<bold>10</bold>
</xref>, and the KPI values are given in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Planned path in environment A. <bold>(a)</bold> Compared with the APF algorithm, <bold>(b)</bold> Compared with the A* algorithm, <bold>(c)</bold> Compared with the DQN algorithm.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g008.tif">
<alt-text content-type="machine-generated">Three grid maps compare pathfinding algorithms by plotting paths around obstacles. Left: APF vs. APF-RADQN; center: A* vs. APF-RADQN; right: DQN vs. APF-RADQN. APF-RADQN paths are consistently longer, shown in red.</alt-text>
</graphic>
</fig>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Planned path in environment B. <bold>(a)</bold> Compared with the APF algorithm, <bold>(b)</bold> Compared with the A* algorithm, <bold>(c)</bold> Compared with the DQN algorithm.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g009.tif">
<alt-text content-type="machine-generated">Three grid-based maps show pathfinding comparisons with obstacles in black. The first map compares APF and APF-RADQN algorithms with cyan and red paths, respectively. The second map compares A* and APF-RADQN algorithms with green and red paths. The third map compares DQN and APF-RADQN algorithms with blue dashed and red paths. Paths navigate around obstacles, illustrating algorithmic differences.</alt-text>
</graphic>
</fig>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Planned path in environment C. <bold>(a)</bold> Compared with the APF algorithm, <bold>(b)</bold> Compared with the A* algorithm, <bold>(c)</bold> Compared with the DQN algorithm.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g010.tif">
<alt-text content-type="machine-generated">Three grid charts compare algorithms for pathfinding among obstacles. The first chart shows paths by APF and APF-RADQN; the second compares A Star and APF-RADQN; the third contrasts DQN and APF-RADQN. Paths navigate black obstacles on a grid, showing variations in trajectory by each algorithm.</alt-text>
</graphic>
</fig>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Algorithmic KPIs.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" rowspan="2" align="left">Environment No.</th>
<th valign="middle" rowspan="2" align="left">Algorithm name</th>
<th valign="middle" colspan="3" align="left">KPIs</th>
</tr>
<tr>
<th valign="middle" align="left">Average path length (m)</th>
<th valign="middle" align="left">Number of corners</th>
<th valign="middle" align="left">Inference times (s)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" rowspan="4" align="left">Env.1</td>
<td valign="middle" align="left">APF-RADQN</td>
<td valign="middle" align="left">96.55</td>
<td valign="middle" align="left">6</td>
<td valign="middle" align="left">0.261</td>
</tr>
<tr>
<td valign="middle" align="left">DQN</td>
<td valign="middle" align="left">99.28</td>
<td valign="middle" align="left">7</td>
<td valign="middle" align="left">0.239</td>
</tr>
<tr>
<td valign="middle" align="left">APF</td>
<td valign="middle" align="left">102.11</td>
<td valign="middle" align="left">10</td>
<td valign="middle" align="left">0.814</td>
</tr>
<tr>
<td valign="middle" align="left">A*</td>
<td valign="middle" align="left">100.98</td>
<td valign="middle" align="left">9</td>
<td valign="middle" align="left">1.484</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="left">Env.2</td>
<td valign="middle" align="left">APF-RADQN</td>
<td valign="middle" align="left">75.04</td>
<td valign="middle" align="left">4</td>
<td valign="middle" align="left">0.154</td>
</tr>
<tr>
<td valign="middle" align="left">DQN</td>
<td valign="middle" align="left">76.5</td>
<td valign="middle" align="left">4</td>
<td valign="middle" align="left">0.136</td>
</tr>
<tr>
<td valign="middle" align="left">APF</td>
<td valign="middle" align="left">82.04</td>
<td valign="middle" align="left">5</td>
<td valign="middle" align="left">0.612</td>
</tr>
<tr>
<td valign="middle" align="left">A*</td>
<td valign="middle" align="left">78.03</td>
<td valign="middle" align="left">4</td>
<td valign="middle" align="left">0.775</td>
</tr>
<tr>
<td valign="middle" rowspan="4" align="left">Env.3</td>
<td valign="middle" align="left">APF-RADQN</td>
<td valign="middle" align="left">74.48</td>
<td valign="middle" align="left">3</td>
<td valign="middle" align="left">0.282</td>
</tr>
<tr>
<td valign="middle" align="left">DQN</td>
<td valign="middle" align="left">78.21</td>
<td valign="middle" align="left">4</td>
<td valign="middle" align="left">0.277</td>
</tr>
<tr>
<td valign="middle" align="left">APF</td>
<td valign="middle" align="left">80.44</td>
<td valign="middle" align="left">8</td>
<td valign="middle" align="left">1.78</td>
</tr>
<tr>
<td valign="middle" align="left">A*</td>
<td valign="middle" align="left">83.38</td>
<td valign="middle" align="left">6</td>
<td valign="middle" align="left">1.91</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>Path analyses of environment A</title>
<p>In <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, the distance between SP and TP is relatively long in environment A. Trap areas do not easily impede the agent. All four algorithms can find collision-free routes. However, there are some differences in details, where the APF algorithm requires the most steps and the A* algorithm has the greatest risk of collision.</p>
<p>In the APF algorithm, shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8a</bold>
</xref>, the target point continuously attracts the agent, while the agent experiences a repulsive force only near obstacles. This mechanism explains why the path generated by the APF algorithm is closer to obstacles than the APF-RADQN generated one during initial navigation phases. With the approaching of the target point, the gravitation force is declared. Thus, the agent in the APF algorithm stays away from obstacles gradually. The comparison with the A* algorithm is shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8b</bold>
</xref>. Because the A* algorithm is only searching nearby points at each point, the path planned by the A* algorithm can&#x2019;t maintain enough safety distance to obstacles when the agent passes through obstacles, as shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8c</bold>
</xref>. The two paths generated by the DQN and APF-RADQN algorithms are similar in nature. However, the APF-RADQN algorithm can generate a smoother path.</p>
</sec>
<sec id="s4_2_2">
<label>4.2.2</label>
<title>Path analyses of environment B</title>
<p>In <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>, the distance between SP and TP is relatively short in environment B. Trap areas do not easily impede the agent. The obstacles in the upper left map are distributed symmetrically. In this environment, all four algorithms can generate the collision-free path to reach the target point, while the APF algorithm generates the longest path, and the A* algorithm has the worst smoothness.</p>
<p>The symmetrical zone makes the APF algorithm set a much broader repulsive force range than usual in case of falling into zero potential energy points. But these parameters setting leads to the longest path in this environment, as shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9a</bold>
</xref>. The other three algorithms have a similar beginning path, but the A* algorithm&#x2019;s planned path is quite near obstacles because of its algorithm searching method (<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9b</bold>
</xref>), and the DQN algorithm generated some unnecessary turns near the target point (<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9c</bold>
</xref>).</p>
</sec>
<sec id="s4_2_3">
<label>4.2.3</label>
<title>Path analyses of environment C</title>
<p>A more complex environment than the previous two maps is shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>. The distance between SP and TP is relatively short in environment C. The obstacles around the SP are formed as a trap that easily impedes the agent. In this environment, all four algorithms can generate a collision-free path to reach the target point. However, the A* algorithm fell into the trap and had the longest path.</p>
<p>The agent&#x2019;s movements at the start need to move in the opposite direction to the target point. This detour examines the interface ability of all four algorithms. As shown in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10a</bold>
</xref>, the APF algorithm can find a way to get out of the initial obstacle because of the combined force. However, multiple obstacles complicate the potential field, resulting in a lot of meaningless turns in the path. <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10b</bold>
</xref> shows the dilemma of the A* algorithm at the early stage. Due to the research method of the A* algorithm, the agent cannot foresee the block between the target point and itself. This disadvantage makes the agent get into the trap at the very beginning. Both the DQN algorithm and the APF-RADQN algorithm show outstanding interface ability in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10c</bold>
</xref>. However, the DQN algorithm is still taking more turns than the APF-RADQN algorithm.</p>
</sec>
<sec id="s4_2_4">
<label>4.2.4</label>
<title>Convergence analyses</title>
<p>
<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> shows the training return curves of the DQN and APF-RADQN algorithms in three different environments. The action-valued network of the agent has not been sufficiently trained in the early stage of training, and the derived strategies are not reasonable enough, so the return curves fluctuate greatly. Subsequently, the USV trains the neural network using the empirical knowledge gained from exploring the environment, then the output of the neural network is gradually close to the real value, and finally, a more reasonable strategy is obtained, and the reward curves are stabilized. Since the APF-RADQN algorithm proposed introduces reward averaging to improve the network error, fully utilizes the rewards from historical experience, and balances the error brought by accidental exploration, the return curve can converge faster in the early stage of training. As the complexity of the environment increases, both DQN and APF-RADQN need more exploration and training to learn the optimal strategy and achieve return curve stabilization. Although the DQN algorithm is able to achieve return curve stabilization after a period of continuous learning, the overestimation of the Q-value leads to slower convergence. By comparing the trend of return values of the proposed APF-RADQN and DQN, it can be seen that the number of iterative steps required for the APF-RADQN algorithm to reach stable reward values in three different environments is reduced by about 7.2%, 13.8% and 10.1%, respectively. Compared with the DQN algorithm, the comparison results show that APF-RADQN has higher learning efficiency, global search capability and excellent convergence.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Return curves. <bold>(a)</bold> Return in environment A, <bold>(b)</bold> Return in environment B, <bold>(c)</bold> Return in environment C.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1641093-g011.tif">
<alt-text content-type="machine-generated">Three line graphs showing return versus episodes in environments A, B, and C. Each graph compares APF-RADQN (blue) and DQN (red) performance. Returns stabilize after initial fluctuations.</alt-text>
</graphic>
</fig>
</sec>
</sec>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusion</title>
<p>In this paper, an algorithm based on the APF-RADQN framework, used for ocean observation missions of USVs, is proposed to address the USV path planning problem. First, a comprehensive reward function combined with the APF algorithm is designed after establishing the USV path planning MDP models, which guides the USV to reach the target area quickly while ensuring that the USV maintains a safe distance from the obstacles. Then, within the framework of the APF-RADQN, a Q-value error calculation method is incorporated, which improves the overall convergence speed and solving capability of the algorithm. In addition, a Bezier curve algorithm is introduced to smooth the discrete movements of the proposed path planning algorithm. Finally, the APF-RADQN-based algorithm is validated in three environments. Compared with DQN, APF algorithm and A* algorithm, the proposed algorithm can find shorter and safer paths and enhance the efficiency of ocean observation missions.</p>
<p>Nevertheless, the proposed algorithm still has some limitations. The algorithm mainly considered the static environments in ocean observation missions, which less consider the dynamic environments and environmental factors such as wind and current. These limitations may influence its usage in complex marine environments. Moreover, the computation burden of the proposed algorithm is quite high, leading to a long interface time. The Future work will focus on improving the algorithm performance in complex ocean observation missions, including using real-time sensor detection to avoid dynamic obstacles, path planning under uncertain environmental perturbations, and the validation of practical algorithms in hardware-closed-loop experiments.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>JM: Writing &#x2013; review &amp; editing, Project administration, Funding acquisition, Methodology. BS: Methodology, Software, Conceptualization, Writing &#x2013; original draft. BW: Validation, Writing &#x2013; review &amp; editing, Investigation. CY: Software, Writing &#x2013; original draft, Methodology. YW: Validation, Writing &#x2013; review &amp; editing, Funding acquisition. FZ: Writing &#x2013; review &amp; editing, Formal Analysis, Software. LZ: Data curation, Writing &#x2013; review &amp; editing, Investigation. JW: Data curation, Writing &#x2013; review &amp; editing, Investigation. JL: Methodology, Writing &#x2013; review &amp; editing, Formal Analysis, Conceptualization.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was supported by the National Natural Science Foundation of China (Grant No. 52175487), the Shandong Province Natural Science Foundation (Grant No. ZR2021ME223), Shandong Offshore Engineering Facility &amp; Material Innovation Entrepreneurship Community (SOFM-IEC) Project (No. GTP-2404), the National Natural Science Foundation of China (Grant No. 52405072).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Author CY was employed by Suzhou Tongyuan Software &amp; Control Technology Co., Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Kamil</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Szabolcsi</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Enhancing mobile robot navigation: optimization of trajectories through machine learning techniques for improved path planning efficiency</article-title>. <source>Mathematics</source> <volume>12</volume>, <elocation-id>1787</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/math12121787</pub-id>
</citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Antonakis</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Nikolaidis</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Pilidis</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Multi-objective climb path optimization for aircraft/engine integration using particle swarm optimization</article-title>. <source>Appl. Sci.</source> <volume>7</volume>, <elocation-id>469</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app7050469</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Hua</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Shuai</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Lane change trajectory prediction considering driving style uncertainty for autonomous vehicles</article-title>. <source>Mechanical Syst. Signal Process.</source> <volume>206</volume>, <elocation-id>110854</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ymssp.2023.110854</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chiodi</surname> <given-names>A. M.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Cokelet</surname> <given-names>E. D.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Mordy</surname> <given-names>C. W.</given-names>
</name>
<name>
<surname>Gentemann</surname> <given-names>C. L.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Exploring the pacific arctic seasonal ice zone with saildrone USVs</article-title>. <source>Front. Mar. Sci.</source> <volume>8</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmars.2021.640697</pub-id>
</citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Lei</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Path planning based on deep reinforcement learning for autonomous underwater vehicles under ocean current disturbance</article-title>. <source>IEEE Trans. Intell. Veh.</source> <volume>8</volume>, <fpage>108</fpage>&#x2013;<lpage>120</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIV.2022.3153352</pub-id>
</citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Moor</surname> <given-names>B. J.</given-names>
</name>
<name>
<surname>Gijsbrechts</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Boute</surname> <given-names>R. N.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Reward shaping to improve the performance of deep reinforcement learning in perishable inventory management</article-title>. <source>Eur. J. Operational Res.</source> <volume>301</volume>, <fpage>535</fpage>&#x2013;<lpage>545</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ejor.2021.10.045</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hadi</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Khosravi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sarhadi</surname> <given-names>P.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Deep reinforcement learning for adaptive path planning and control of an autonomous underwater vehicle</article-title>. <source>Appl. Ocean Res.</source> <volume>129</volume>, <elocation-id>103326</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.apor.2022.103326</pub-id>
</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Handegard</surname> <given-names>N. O.</given-names>
</name>
<name>
<surname>De Robertis</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Holmin</surname> <given-names>A. J.</given-names>
</name>
<name>
<surname>Johnsen</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Lawrence</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Le Bouffant</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Uncrewed surface vehicles (USVs) as platforms for fisheries and plankton acoustics</article-title>. <source>ICES J. Mar. Sci.</source> <volume>81</volume>, <fpage>1712</fpage>&#x2013;<lpage>1723</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/icesjms/fsae130</pub-id>
</citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Fuzzy A&#x2217; quantum multi-stage Q-learning artificial potential field for path planning of mobile robots</article-title>. <source>Eng. Appl. Artif. Intell.</source> <volume>141</volume>, <elocation-id>109866</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.engappai.2024.109866</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ionescu</surname> <given-names>T. B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Adaptive simplex architecture for safe, real-time robot path planning</article-title>. <source>Sensors</source> <volume>21</volume>, <elocation-id>2589</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s21082589</pub-id>, PMID: <pub-id pub-id-type="pmid">33917089</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Mining trajectory planning of unmanned excavator based on machine learning</article-title>. <source>Mathematics</source> <volume>12</volume>, <elocation-id>1298</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/math12091298</pub-id>
</citation></ref>
<ref id="B12">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Khatib</surname> <given-names>O.</given-names>
</name>
</person-group> (<year>1985</year>). &#x201c;<article-title>Real-time obstacle avoidance for manipulators and mobile robots</article-title>,&#x201d; in <conf-name>Proceedings. 1985 IEEE International Conference on Robotics and Automation</conf-name>, (<publisher-loc>St. Louis, MO, USA</publisher-loc>: <publisher-name>IEEE</publisher-name>). <volume>2</volume>. <fpage>500</fpage>&#x2013;<lpage>505</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ROBOT.1985.1087247</pub-id>
</citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koval</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Karlsson</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Nikolakopoulos</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Experimental evaluation of autonomous map-based Spot navigation in confined environments</article-title>. <source>Biomimetic Intell. Robotics</source> <volume>2</volume>, <elocation-id>100035</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.birob.2022.100035</pub-id>
</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lan</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Tian</surname> <given-names>W.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Path planning for underwater gliders in time-varying ocean current using deep reinforcement learning</article-title>. <source>Ocean Eng.</source> <volume>262</volume>, <elocation-id>112226</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.oceaneng.2022.112226</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yue</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2025</year>). <article-title>A modified dueling DQN algorithm for robot path planning incorporating priority experience replay and artificial potential fields</article-title>. <source>Appl. Intell.</source> <volume>55</volume>, <fpage>366</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10489-024-06149-8</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>Z.-M.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A path planning strategy unified with a COLREGS collision avoidance function based on deep reinforcement learning and artificial potential field</article-title>. <source>Appl. Ocean Res.</source> <volume>113</volume>, <elocation-id>102759</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.apor.2021.102759</pub-id>
</citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Distributed formation control using artificial potentials and neural network for constrained multiagent systems</article-title>. <source>IEEE Trans. Contr. Syst. Technol.</source> <volume>28</volume>, <fpage>697</fpage>&#x2013;<lpage>704</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TCST.2018.2884226</pub-id>
</citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Low</surname> <given-names>E. S.</given-names>
</name>
<name>
<surname>Ong</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Cheah</surname> <given-names>K. C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Solving the optimal path planning of a mobile robot using improved Q-learning</article-title>. <source>Robotics Autonomous Syst.</source> <volume>115</volume>, <fpage>143</fpage>&#x2013;<lpage>161</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.robot.2019.02.013</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Bi</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Kr&#xf3;lczyk</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A new coverage path planning algorithm for unmanned surface mapping vehicle based on A-star based searching</article-title>. <source>Appl. Ocean Res.</source> <volume>123</volume>, <elocation-id>103163</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.apor.2022.103163</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mnih</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Kavukcuoglu</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Silver</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Rusu</surname> <given-names>A. A.</given-names>
</name>
<name>
<surname>Veness</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Bellemare</surname> <given-names>M. G.</given-names>
</name>
<etal/>
</person-group>. (<year>2015</year>). <article-title>Human-level control through deep reinforcement learning</article-title>. <source>Nature</source> <volume>518</volume>, <fpage>529</fpage>&#x2013;<lpage>533</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature14236</pub-id>, PMID: <pub-id pub-id-type="pmid">25719670</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Naik</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shariff</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Yasui</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Sutton</surname> <given-names>R. S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Discounted reinforcement learning is not an optimization problem</article-title>. <source>arXiv [prepint]</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1910.02140</pub-id>
</citation></ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pflueger</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Agha</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sukhatme</surname> <given-names>G. S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Rover-IRL: inverse reinforcement learning with soft value iteration networks for planetary rover path planning</article-title>. <source>IEEE Robot. Autom. Lett.</source> <volume>4</volume>, <fpage>1387</fpage>&#x2013;<lpage>1394</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/LRA.2019.2895892</pub-id>
</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Santos</surname> <given-names>R. R.</given-names>
</name>
<name>
<surname>Rade</surname> <given-names>D. A.</given-names>
</name>
<name>
<surname>Da Fonseca</surname> <given-names>I. M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A machine learning strategy for optimal path planning of space robotic manipulator in on-orbit servicing</article-title>. <source>Acta Astronautica</source> <volume>191</volume>, <fpage>41</fpage>&#x2013;<lpage>54</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.actaastro.2021.10.031</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Jiang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Tong</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Path planning optimization of intelligent vehicle based on improved genetic and ant colony hybrid algorithm</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>10</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fbioe.2022.905983</pub-id>, PMID: <pub-id pub-id-type="pmid">35845413</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>An improved PSO algorithm for smooth path planning of mobile robots using continuous high-degree Bezier curve</article-title>. <source>Appl. Soft Computing</source> <volume>100</volume>, <elocation-id>106960</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.asoc.2020.106960</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Su</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J.-B.</given-names>
</name>
<name>
<surname>Zeng</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>G. Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Unmanned-surface-vehicle-aided maritime data collection using deep reinforcement learning</article-title>. <source>IEEE Internet Things J.</source> <volume>9</volume>, <fpage>19773</fpage>&#x2013;<lpage>19786</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JIOT.2022.3168589</pub-id>
</citation></ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sutton</surname> <given-names>R. S.</given-names>
</name>
<name>
<surname>Barto</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Reinforcement learning: an introduction</source>. <edition>2nd ed.</edition> (<publisher-loc>Cambridge, Massachusetts London, England</publisher-loc>: <publisher-name>The MIT Press</publisher-name>).</citation></ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tan</surname> <given-names>C. S.</given-names>
</name>
<name>
<surname>Mohd-Mokhtar</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Arshad</surname> <given-names>M. R.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Expected-mean gamma-incremental reinforcement learning algorithm for robot path planning</article-title>. <source>Expert Syst. Appl.</source> <volume>249</volume>, <elocation-id>123539</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2024.123539</pub-id>
</citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tsai</surname> <given-names>C.-C.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>H.-C.</given-names>
</name>
<name>
<surname>Chan</surname> <given-names>C.-K.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Parallel elite genetic algorithm and its application to global path planning for autonomous robot navigation</article-title>. <source>IEEE Trans. Ind. Electron.</source> <volume>58</volume>, <fpage>4813</fpage>&#x2013;<lpage>4821</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIE.2011.2109332</pub-id>
</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Jing</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>He</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Qi</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Lou</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>c). <article-title>ETQ-learning: an improved Q-learning algorithm for path planning</article-title>. <source>Intel Serv. Robotics</source> <volume>17</volume>, <fpage>915</fpage>&#x2013;<lpage>929</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11370-024-00544-3</pub-id>
</citation></ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Prorok</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Mobile robot path planning in dynamic environments through globally guided reinforcement learning</article-title>. <source>IEEE Robot. Autom. Lett.</source> <volume>5</volume>, <fpage>6932</fpage>&#x2013;<lpage>6939</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/LRA.2020.3026638</pub-id>
</citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Mao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Mou</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2025</year>a). <article-title>Path planning for unmanned surface vehicles in anchorage areas based on the risk-aware path optimization algorithm</article-title>. <source>Front. Mar. Sci.</source> <volume>11</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmars.2024.1503482</pub-id>
</citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2025</year>b). <article-title>Path planning of mobile robot based on improved double deep Q-network algorithm</article-title>. <source>Front. Neurorobot.</source> <volume>19</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fnbot.2025.1512953</pub-id>, PMID: <pub-id pub-id-type="pmid">40018324</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Bashir</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>a). <article-title>COLERGs-constrained safe reinforcement learning for realising MASS&#x2019;s risk-informed collision avoidance decision making</article-title>. <source>Knowledge-Based Syst.</source> <volume>300</volume>, <elocation-id>112205</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.knosys.2024.112205</pub-id>
</citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Bashir</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>b). <article-title>Optimizing anti-collision strategy for MASS: A safe reinforcement learning approach to improve maritime traffic safety</article-title>. <source>Ocean Coast. Manage.</source> <volume>253</volume>, <elocation-id>107161</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ocecoaman.2024.107161</pub-id>
</citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Manfredi</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Path planning for active SLAM based on deep reinforcement learning under unknown environments</article-title>. <source>Intel Serv. Robotics</source> <volume>13</volume>, <fpage>263</fpage>&#x2013;<lpage>272</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11370-019-00310-w</pub-id>
</citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wills</surname> <given-names>S. M.</given-names>
</name>
<name>
<surname>Cronin</surname> <given-names>M. F.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Air-sea heat fluxes associated with convective cold pools</article-title>. <source>JGR Atmospheres</source> <volume>128</volume>, <elocation-id>e2023JD039708</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1029/2023JD039708</pub-id>
</citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Mao</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Cooperative path planning for multiple UAVs based on APF B-RRT* Algorithm</article-title>. <source>Drones</source> <volume>9</volume>, <elocation-id>177</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/drones9030177</pub-id>
</citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Convolutionally evaluated gradient first search path planning algorithm without prior global maps</article-title>. <source>Robotics Autonomous Syst.</source> <volume>150</volume>, <elocation-id>103985</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.robot.2021.103985</pub-id>
</citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>H. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Comprehensive ocean information-enabled AUV path planning via reinforcement learning</article-title>. <source>IEEE Internet Things J.</source> <volume>9</volume>, <fpage>17440</fpage>&#x2013;<lpage>17451</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JIOT.2022.3155697</pub-id>
</citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xiaofei</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Yilun</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Hui</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Weibo</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhengrong</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Global path planning algorithm based on double DQN for multi-tasks amphibious unmanned surface vehicle</article-title>. <source>Ocean Eng.</source> <volume>266</volume>, <elocation-id>112809</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.oceaneng.2022.112809</pub-id>
</citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Ni</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xi</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Intelligent path planning of underwater robot based on reinforcement learning</article-title>. <source>IEEE Trans. Automat. Sci. Eng.</source> <volume>20</volume>, <fpage>1983</fpage>&#x2013;<lpage>1996</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TASE.2022.3190901</pub-id>
</citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>W.-N.</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>ACO-A*: ant colony optimization plus A* for 3-D traveling in environments with dense obstacles</article-title>. <source>IEEE Trans. Evol. Computat.</source> <volume>23</volume>, <fpage>617</fpage>&#x2013;<lpage>631</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TEVC.2018.2878221</pub-id>
</citation></ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>UAV path design with connectivity constraint based on deep reinforcement learning</article-title>. <source>Phys. Communication</source> <volume>52</volume>, <elocation-id>101582</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.phycom.2021.101582</pub-id>
</citation></ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Gai</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Double-DQN based path smoothing and tracking control method for robotic vehicle navigation</article-title>. <source>Comput. Electron. Agric.</source> <volume>166</volume>, <elocation-id>104985</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.compag.2019.104985</pub-id>
</citation></ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Bai</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Data harvesting in uncharted waters: Interactive learning empowered path planning for USV-assisted maritime data collection under fully unknown environments</article-title>. <source>Ocean Eng.</source> <volume>287</volume>, <elocation-id>115781</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.oceaneng.2023.115781</pub-id>
</citation></ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Bachmayer</surname> <given-names>R.</given-names>
</name>
<name>
<surname>DeYoung</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Surveying a floating iceberg with the USV SEADRAGON</article-title>. <source>Front. Mar. Sci.</source> <volume>8</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmars.2021.549566</pub-id>
</citation></ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Sheng</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>An improved dueling deep double-Q network based on prioritized experience replay for path planning of unmanned surface vehicles</article-title>. <source>JMSE</source> <volume>9</volume>, <elocation-id>1267</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/jmse9111267</pub-id>
</citation></ref>
</ref-list>
</back>
</article>