<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1492526</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1492526</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Reinforcement learning-based dynamic field exploration and reconstruction using multi-robot systems for environmental monitoring</article-title>
<alt-title alt-title-type="left-running-head">Lu et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1492526">10.3389/frobt.2025.1492526</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Lu</surname>
<given-names>Thinh</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2835936/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sobti</surname>
<given-names>Divyam</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/2874944/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Talwar</surname>
<given-names>Deepak</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3000841/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wu</surname>
<given-names>Wencen</given-names>
</name>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1100114/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
</contrib-group>
<aff>
<institution>Computer Engineering Department</institution>, <institution>San Jose State University</institution>, <addr-line>San Jose</addr-line>, <addr-line>CA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1848490/overview">Shude He</ext-link>, Guangzhou University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2854535/overview">Ke Lu</ext-link>, Guangxi University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2861376/overview">Lin Chen</ext-link>, Hunan University, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Wencen Wu, <email>wencen.wu@sjsu.edu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>25</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1492526</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>09</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Lu, Sobti, Talwar and Wu.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Lu, Sobti, Talwar and Wu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>In the realm of real-time environmental monitoring and hazard detection, multi-robot systems present a promising solution for exploring and mapping dynamic fields, particularly in scenarios where human intervention poses safety risks. This research introduces a strategy for path planning and control of a group of mobile sensing robots to efficiently explore and reconstruct a dynamic field consisting of multiple non-overlapping diffusion sources. Our approach integrates a reinforcement learning-based path planning algorithm to guide the multi-robot formation in identifying diffusion sources, with a clustering-based method for destination selection once a new source is detected, to enhance coverage and accelerate exploration in unknown environments. Simulation results and real-world laboratory experiments demonstrate the effectiveness of our approach in exploring and reconstructing dynamic fields. This study advances the field of multi-robot systems in environmental monitoring and has practical implications for rescue missions and field explorations.</p>
</abstract>
<kwd-group>
<kwd>multi-robot systems</kwd>
<kwd>mobile sensor networks</kwd>
<kwd>reinforcement learning</kwd>
<kwd>dynamic field reconstruction</kwd>
<kwd>source seeking</kwd>
<kwd>environmental monitoring</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Robot Learning and Evolution</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Environmental monitoring, including the identification and tracing of areas impacted by environmental hazards, is paramount for safeguarding human life and property. Early warning systems allow for swift responses to potential threats. Effective environmental monitoring relies on a deep understanding of key processes like wildfire propagation and pollutant dispersion. These phenomena often involve spatial and temporal changes, making them suitable for modeling using partial differential equations (PDEs). For instance, the advection-diffusion equation can be used to simulate the movement of smoke plumes from wildfires, providing crucial insights for predicting their evolution over time (<xref ref-type="bibr" rid="B13">Khaled et al., 2004</xref>; <xref ref-type="bibr" rid="B26">Reisch et al., 2024</xref>). This information is essential for effective environmental hazard monitoring and mitigation.</p>
<p>For environmental monitoring tasks, multi-robot systems offer significant advantages over single-robot setups by enabling faster coverage of larger areas and providing redundancy against individual failures. These systems excel in complex missions across diverse environments, including search and rescue (<xref ref-type="bibr" rid="B21">Niroui et al., 2019</xref>; <xref ref-type="bibr" rid="B32">Shuvo et al., 2023</xref>; <xref ref-type="bibr" rid="B4">Cao et al., 2024</xref>), underwater surveillance (<xref ref-type="bibr" rid="B18">Martins et al., 2018</xref>; <xref ref-type="bibr" rid="B17">Luvisutto et al., 2022</xref>), and space exploration (<xref ref-type="bibr" rid="B10">Gautam et al., 2019</xref>; <xref ref-type="bibr" rid="B2">Bi et al., 2024</xref>; <xref ref-type="bibr" rid="B16">Long and Zhang, 2024</xref>). These works place additional emphasis on developing robust coordination strategies and efficient path-planning algorithms. Coordination may be centralized, with a leader directing actions, or decentralized, with robots making their own decisions based on local observations. Reinforcement learning has advanced these strategies, with actor&#x2013;critic models enhancing control stability of the whole unit under dynamic disturbances (<xref ref-type="bibr" rid="B11">Hu et al., 2023</xref>) and graph-based methods enabling scalable, distributed decision-making across large robot teams (<xref ref-type="bibr" rid="B5">Chen et al., 2024</xref>). Depending on the mission, formation control may also play an essential role, where the robot system can be operated in organized patterns for high-quality data collection, or independently for greater flexibility. Beyond coordination, reinforcement learning-based approaches have also been increasingly adopted for path planning, further enhancing adaptability and performance of multi-robot systems in unknown environments (<xref ref-type="bibr" rid="B47">Zhu et al., 2023</xref>). These approaches require careful design of both the simulation environment and reward functions, which should closely model real-world conditions, to ensure effective learning and reliable performance in deployment. For applications in environmental monitoring, multi-robot systems can be equipped with specialized sensors to enable real-time data collection and reconstruction of environmental processes (<xref ref-type="bibr" rid="B14">Kinaneva et al., 2019</xref>; <xref ref-type="bibr" rid="B9">Dunbabin and Marques, 2012</xref>; <xref ref-type="bibr" rid="B23">Queralta et al., 2020</xref>; <xref ref-type="bibr" rid="B29">Rossi and Brunelli, 2016</xref>).</p>
<p>To reconstruct dynamic processes through limited measurements from multi-robot systems, it is necessary to identify unknown parameters in the PDEs that describe these processes, such as the diffusion coefficient in a diffusion equation. A common approach is to deploy static sensor networks (<xref ref-type="bibr" rid="B19">Mourikis and Roumeliotis, 2006</xref>; <xref ref-type="bibr" rid="B3">Burgard et al., 2005</xref>). Although effective, this approach is both costly and impractical for large-scale regions due to the need for extensive sensor installations. Mobile sensor networks, with collaborative mobile sensing robots, present a more practical alternative, offering great flexibility and broad coverage while using fewer sensors. In mobile sensor networks, parameter identification can be performed in two primary ways: offline and online (<xref ref-type="bibr" rid="B46">Zhang et al., 2023</xref>). Offline parameter identification requires mobile sensor networks to explore the entire spatial domain before any parameter estimation begins (<xref ref-type="bibr" rid="B35">Ucinski, 2005</xref>; <xref ref-type="bibr" rid="B36">Ucinski and Chen, 2005</xref>; <xref ref-type="bibr" rid="B34">Tricaud and Chen, 2010</xref>). This approach often uses techniques like least squares optimization to minimize the error between the observed and estimated states, typically requiring complex computations to solve PDEs. While this approach generally yields more accurate results, it is time-consuming, as full data collection must be completed before any estimation can take place. Due to the limitations of offline methods, increasing attention has shifted toward online parameter identification approaches (<xref ref-type="bibr" rid="B39">Wu et al., 2020</xref>; <xref ref-type="bibr" rid="B46">Zhang et al., 2023</xref>; <xref ref-type="bibr" rid="B6">Christopoulos and Roumeliotis, 2005</xref>). Online identification continuously updates parameter estimates as mobile sensors collect data in real time. While this approach may not provide the most accurate solution to PDEs compared to offline methods, it is far more efficient for time-sensitive applications like environmental hazard management (<xref ref-type="bibr" rid="B46">Zhang et al., 2023</xref>).</p>
<p>A key challenge of online parameter identification is determining an information-rich trajectory for the mobile sensing network, as this directly impacts the speed and accuracy of field reconstruction. However, since online methods operate in real-time, predicting the optimal path in advance is challenging, making efficient trajectory planning a complex problem. As a result, recent works in this field often provide additional strategies for effective trajectory planning and navigation for mobile sensor networks. In (<xref ref-type="bibr" rid="B43">You et al., 2016</xref>; <xref ref-type="bibr" rid="B46">Zhang et al., 2023</xref>), the authors employ a cooperative Kalman filter (CKF) combined with recursive least squares (RLS) to identify advection-diffusion field parameters in real-time using live sensor readings from a formation of mobile robots. To ensure that the robot formation follows information-rich trajectories, several studies, including (<xref ref-type="bibr" rid="B42">You and Wu, 2018</xref>; <xref ref-type="bibr" rid="B44">You et al., 2022</xref>), have integrated robot dynamics into the field dynamics and focused on minimizing mapping errors. However, these approaches may converge to local optima and may not adequately address the complexity of field reconstruction in environments involving multiple diffusion fields with varying characteristics. To address this issue, The author in (<xref ref-type="bibr" rid="B33">Talwar, 2020</xref>) proposes an exploration strategy that samples nearby candidate destinations based on custom weights calculated from cosine similarity to the centroid of unvisited regions and distance from explored diffusion fields. However, this approach may result in inefficient backtracking and revisits, which are undesirable in time-critical missions.</p>
<p>To tackle the problem of exploring complex dynamic fields, this research introduces a strategy for path planning and control of mobile sensing robots designed to effectively explore and reconstruct a dynamic field consisting of multiple non-overlapping diffusion fields while offering a good balance between speed and accuracy. In our proposed algorithm, the robot formation alternates between two primary operational modes: Field Exploration and Source Mapping. In Source Mapping mode, the formation makes use of reinforcement learning (RL), specifically, proximal policy optimization (PPO) to direct the robot formation to the center of a newly discovered diffusion field, while attempting to estimate its diffusion and advection coefficients through the CKF and RLS developed in (<xref ref-type="bibr" rid="B44">You et al., 2022</xref>). When dealing with the challenging problem that multiple sources exist in the field and the path planned in Source Mapping mode only leads to one source (local maximum) in the field, we develop a novel K-means clustering algorithm in the Field Exploration mode, to allow the robot formation advances toward unexplored regions to identify traces of potential new diffusion fields. The K-means clustering algorithm is used to partition the unexplored regions and facilitate faster scanning of the whole map. We validate our proposed strategy through both computer simulations and controlled laboratory experiments. In these scenarios, the robot formation is randomly placed within a spatially and temporally varying field, and we compare the field reconstruction errors to baseline strategies that employ random or lawn-mowing trajectories. Our research demonstrates the potential of multi-robot formations for accurate field reconstruction in complex environments characterized by multiple spatial-temporal diffusion fields.</p>
<p>To summarize, the main contributions of the paper are twofold: (1) it introduces a novel two-mode strategy for path planning and control of mobile sensing robots in dynamic environments, specifically for exploring and reconstructing fields with multiple non-overlapping diffusion sources. The strategy integrates RL-based path planning with a CKF and RLS for estimating unknown parameters of the field. A key innovation is the use of K-means clustering algorithm to facilitate efficient exploration of unexplored regions, ensuring a balance between speed and accuracy. (2) Through both simulations and controlled experiments, the research demonstrates the effectiveness of the proposed strategy in improving field reconstruction accuracy using only a limited number of mobile sensing robots.</p>
<p>The remainder of this paper is structured as follows. In <xref ref-type="sec" rid="s2">Section 2</xref>, we formally define the problem. We present some preliminary information in <xref ref-type="sec" rid="s3">Section 3</xref>. The proposed algorithm is presented in <xref ref-type="sec" rid="s4">Section 4</xref>, followed by a detailed analysis of the simulation and experimental results in <xref ref-type="sec" rid="s5">Section 5</xref> and <xref ref-type="sec" rid="s6">Section 6</xref> respectively. Finally, <xref ref-type="sec" rid="s7">Section 7</xref> summarizes our findings and outlines future research directions.</p>
</sec>
<sec id="s2">
<title>2 Problem formulation</title>
<p>In this section, we formulate the problem of reconstructing an unknown spatial-temporal varying field represented by a linear combination of several advection-diffusion equations, using a team of mobile sensing robots.</p>
<sec id="s2-1">
<title>2.1 Spatial-temporal varying fields</title>
<p>Various processes that exhibit spatial and temporal variations, such as the dispersion of pollutants in the atmosphere or water bodies, are often represented by two-dimensional (2D) PDEs over a domain <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. A typical example is the 2D advection-diffusion equation, which models the transfer of substances via advection (the movement of substances through a fluid) and diffusion (the spreading of substances from areas of higher to lower concentration). This can be mathematically expressed as:<disp-formula id="e1">
<mml:math id="m2">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi mathvariant="bold">v</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the concentration function of the field at position <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at time instance <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are the gradient and Laplacian operators, respectively. The coefficient <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> represents the rate of diffusion, and <inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the two-dimensional advection coefficient, representing the speed at which a quantity such as heat, concentration, or pollutant is transported by the bulk movement of a fluid. Both <inline-formula id="inf9">
<mml:math id="m10">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf10">
<mml:math id="m11">
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are considered constant but possibly unknown over a fixed interval.</p>
<p>In this work, we consider the field as a linear superposition of multiple non-overlapping advection diffusion phenomena, each governing a spatial-temporal region <inline-formula id="inf11">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> where <inline-formula id="inf12">
<mml:math id="m13">
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the number of the advection-diffusion processes. The concentration <italic>z<sub>i(r,t)</sub>
</italic> in each region satisfies the advection-diffusion equation: <inline-formula id="inf13">
<mml:math id="m14">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi mathvariant="bold">v</mml:mi>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mi>&#x2207;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x2009;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The global concentration field is then represented as<disp-formula id="e2">
<mml:math id="m15">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:msub>
<mml:mi>&#x3c7;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <italic>&#x3c7;<sub>i</sub>
</italic>(<italic>r</italic>) is an indicator function defined as <inline-formula id="inf14">
<mml:math id="m16">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c7;</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if <inline-formula id="inf15">
<mml:math id="m17">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and 0 otherwise, and <inline-formula id="inf16">
<mml:math id="m18">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x22c3;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Moreover, in various real-world environmental monitoring scenarios, the domain <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is significantly larger than the dimensions of the operational robots, enabling the approximation of the boundary as essentially flat. Under these conditions, we apply the initial and Dirichlet boundary conditions as shown in <xref ref-type="disp-formula" rid="e3">Equation 3</xref> at the boundary <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:mi>&#x2202;</mml:mi>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B7">Demetriou et al., 2013</xref>):<disp-formula id="e3">
<mml:math id="m21">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>&#x2202;</mml:mi>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-2">
<title>2.2 Mobile sensing robots</title>
<p>In this work, we consider a group of <inline-formula id="inf19">
<mml:math id="m22">
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> mobile sensing robots moving in a coordinated formation in the field <inline-formula id="inf20">
<mml:math id="m23">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represented in <xref ref-type="disp-formula" rid="e2">Equation 2</xref>. The algorithm proposed in this work commands the formation to travel on planned paths to efficiently reconstruct the unknown field. We make the following assumption regarding these mobile sensing robots.</p>
<p>
<statement content-type="assumption" id="Assumption_2_1">
<label>Assumption 2.1</label>
<p>
<italic>Each sensing robot is equipped with sensors to localize itself and to measure the field concentration value at its current location at each discrete time step</italic> <inline-formula id="inf21">
<mml:math id="m24">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</statement>
</p>
<p>The measurement of the <inline-formula id="inf22">
<mml:math id="m25">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th sensing robot at time step <inline-formula id="inf23">
<mml:math id="m26">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is modeled as follows:<disp-formula id="e4">
<mml:math id="m27">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf24">
<mml:math id="m28">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> represents the location of the <inline-formula id="inf25">
<mml:math id="m29">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>th robot at the discrete time step <inline-formula id="inf26">
<mml:math id="m30">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf27">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is assumed to be i.i.d Gaussian noise. Additionally, using the locations of all the robots in the formation at time step <inline-formula id="inf28">
<mml:math id="m32">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, we can determine the location of the formation center <inline-formula id="inf29">
<mml:math id="m33">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> at time step <inline-formula id="inf30">
<mml:math id="m34">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as <inline-formula id="inf31">
<mml:math id="m35">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>When the robots move in a desired formation, it covers a time-varying view-scope <inline-formula id="inf32">
<mml:math id="m36">
<mml:mrow>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, which is the area of the field domain <inline-formula id="inf33">
<mml:math id="m37">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> that lies inside the polygon formed by sensing robot locations. As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, the shaded region illustrates the time-varying view-scope <inline-formula id="inf34">
<mml:math id="m38">
<mml:mrow>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> at discrete time step <inline-formula id="inf35">
<mml:math id="m39">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the blue circles represent the four mobile robots, and the red circle represents the formation center. At any given time, the mobile sensing robots can measure and exchange concentration values at their specific locations as shown in <xref ref-type="disp-formula" rid="e4">Equation 4</xref> and the field values <inline-formula id="inf36">
<mml:math id="m40">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> can be estimated by interpolating the measured values from the robots. Consequently, it is reasonable to assume that the estimated field values, <inline-formula id="inf37">
<mml:math id="m41">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, are available to us at all times.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>A symmetric formation composed of four mobile robots <inline-formula id="inf38">
<mml:math id="m42">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0,1,2,3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> shown in blue. The formation center <inline-formula id="inf39">
<mml:math id="m43">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is shown in red. The distance between each robot and the formation center is <inline-formula id="inf40">
<mml:math id="m44">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The shaded region is the time-varying view-scope <inline-formula id="inf41">
<mml:math id="m45">
<mml:mrow>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g001.tif"/>
</fig>
<p>In this work, to facilitate the implementation of the PPO algorithm for source mapping in <xref ref-type="sec" rid="s4-1">Section 4.1</xref>, we discretize the global field <inline-formula id="inf42">
<mml:math id="m46">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to a <inline-formula id="inf43">
<mml:math id="m47">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> grid, where each grid cell represents a single location <inline-formula id="inf44">
<mml:math id="m48">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> in the map and associates with a concentration value <inline-formula id="inf45">
<mml:math id="m49">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The following assumption holds for the formation center.</p>
<p>
<statement content-type="assumption" id="Assumption_2_2">
<label>Assumption 2.2</label>
<p>
<italic>Robots travel in a coordinate formation and the formation center moves along the eight possible directions &#x201c;up&#x201d;, &#x201c;down&#x201d;, &#x201c;left&#x201d;, &#x201c;right&#x201d;, &#x201c;up-left&#x201d;, &#x201c;up-right&#x201d;, &#x201c;down-left&#x201d;, &#x201c;down-right&#x201d; in the discretized domain.</italic>
</p>
</statement>
</p>
<p>With the robots moving in a formation, a CKF developed in (<xref ref-type="bibr" rid="B43">You et al., 2016</xref>; <xref ref-type="bibr" rid="B39">Wu et al., 2020</xref>) is employed to output estimates of concentration <inline-formula id="inf46">
<mml:math id="m50">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and gradients <inline-formula id="inf47">
<mml:math id="m51">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> at the formation center <inline-formula id="inf48">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> at time step <inline-formula id="inf49">
<mml:math id="m53">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. These estimated values will play a major role in the developed algorithm in <xref ref-type="sec" rid="s4">Section 4</xref>. Furthermore, we apply the parameter identification algorithms developed in (<xref ref-type="bibr" rid="B39">Wu et al., 2020</xref>) to estimate the unknown diffusion coefficient <inline-formula id="inf50">
<mml:math id="m54">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in real-time using the output from the CKF and the RLS algorithm. Thus, in the following discussions, we consider <inline-formula id="inf51">
<mml:math id="m55">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as a known value for field reconstruction.</p>
<p>
<statement content-type="remark" id="Remark_2_1">
<label>Remark 2.1</label>
<p>
<italic>Multi-robot formation control is a well-studied topic and researchers have developed numerous formation control algorithms</italic> (<xref ref-type="bibr" rid="B45">Zhang and Leonard, 2010</xref>; <xref ref-type="bibr" rid="B27">Ren and Beard, 2008</xref>; <xref ref-type="bibr" rid="B40">Wu and Zhang, 2012</xref>)<italic>. In this work, we employ the formation control strategy developed in</italic> (<xref ref-type="bibr" rid="B45">Zhang and Leonard, 2010</xref>) <italic>and applied in</italic> (<xref ref-type="bibr" rid="B39">Wu et al., 2020</xref>)<italic>. The strategy uses the Jacobi transform to decouple the formation control from the motion control of the multi-robot formation, which enables us to only plan the path and design the controller for the formation center</italic> <inline-formula id="inf52">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
<italic>. The individual robot controllers are then derived using the formation controller.</italic>
</p>
</statement>
</p>
</sec>
<sec id="s2-3">
<title>2.3 The multi-robot source seeking and field reconstruction problem</title>
<p>In real-world scenarios, the task of mapping complex dynamic fields for cases like gas-leaking and wildfires is important and is often time-critical. It is essential for the robot formation to explore and detect diffusion sources in unknown areas and generate a map as quickly as possible. With the field defined in <xref ref-type="sec" rid="s2-1">Section 2.1</xref> and the multi-robot formation defined in <xref ref-type="sec" rid="s2-2">Section 2.2</xref>, the goal of this study is to design a path for the multi-robot formation so that the formation can identify the multiple non-overlapping diffusion sources in the dynamic field and reconstruct the field in real-time with the limited concentration measurements collected by the multi-robot formation along its trajectory. To achieve the goal, we will introduce a two-mode strategy in <xref ref-type="sec" rid="s4">Section 4</xref>, which consists of a Source Mapping mode and a Field Exploration mode. In the Source Mapping mode, we employ the RL-based algorithm and train a PPO model to guide the multi-robot formation toward a diffusion source in the field and reach a stationary state, where the formation arrives at the source and moves with the field at the same speed as the advection flow. In the Field Exploration mode, we develop a K-means clustering-based exploration strategy to enable efficient exploration of unknown areas.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Preliminaries</title>
<sec id="s3-1">
<title>3.1 Proximal policy optimization</title>
<p>PPO (<xref ref-type="bibr" rid="B31">Schulman et al., 2017</xref>) is a significant development in reinforcement learning, introduced as a more efficient and simpler alternative to Trust Region Policy Optimization (TRPO) (<xref ref-type="bibr" rid="B30">Schulman et al., 2015</xref>). PPO is based on the policy gradient approach, a class of algorithms that optimize policies by directly computing gradients of expected rewards for policy parameters. This approach allows the learning agent to improve its policy iteratively by following the gradient of expected rewards. PPO enhances this process by addressing the complexities of earlier methods while retaining their benefits, particularly in maintaining stable and reliable policy updates.</p>
<p>PPO operates as an on-policy method. Unlike traditional policy gradient methods that apply a single update after each interaction with the environment, PPO refines the policy by using multiple updates on the same batch of data. The core of PPO is its surrogate objective function, designed to prevent large, potentially destructive policy updates. This is achieved through a probability ratio <inline-formula id="inf53">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> between the new and old policies, which is clipped to keep updates within a safe range. The surrogate objective function <inline-formula id="inf54">
<mml:math id="m58">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">CLIP</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is expressed as:<disp-formula id="e5">
<mml:math id="m59">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo movablelimits="false" form="prefix">clip</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>In <xref ref-type="disp-formula" rid="e5">Equation 5</xref>, <inline-formula id="inf55">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the expectation over timestep <inline-formula id="inf56">
<mml:math id="m61">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf57">
<mml:math id="m62">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represents the policy parameters, <inline-formula id="inf58">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is an estimate of the advantage function at time step <inline-formula id="inf59">
<mml:math id="m64">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf60">
<mml:math id="m65">
<mml:mrow>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a small hyperparameter that controls the clipping range. It is important to note that while <inline-formula id="inf61">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf62">
<mml:math id="m67">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> follow the conventional notations used in literature, they differ from the notations in other sections of this paper, where <inline-formula id="inf63">
<mml:math id="m68">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> refers to the locations in the field and <inline-formula id="inf64">
<mml:math id="m69">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> refers to the diffusion coefficient. The clipping mechanism ensures that if the probability ratio deviates outside the predefined range <inline-formula id="inf65">
<mml:math id="m70">
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, the function applies the clipped values to prevent excessively large updates, thereby maintaining the stability of the learning process. By constraining the probability ratio, PPO effectively controls the size of policy updates, balancing stability and performance. PPO is particularly well-suited for discrete action spaces, which makes it an ideal choice for our environment setup. The PPO algorithm is summarized in <xref ref-type="statement" rid="Algorithm_1">Algorithm 1</xref>.</p>
<p>
<statement content-type="algorithm" id="Algorithm_1">
<label>Algorithm 1</label>
<p>PPO-Clip Algorithm.<list list-type="simple">
<list-item>
<p>
<bold>Input:</bold>&#x2003;Initial policy parameters&#x2003;<inline-formula id="inf66">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, initial value function parameters <inline-formula id="inf67">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>
<bold>for</bold>&#x2003;<inline-formula id="inf68">
<mml:math id="m73">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0,1,2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mtext>&#x2009;</mml:mtext>
</mml:mrow>
</mml:math>
</inline-formula>&#x2003;<bold>do</bold>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Generate a set of trajectories&#x2003;<inline-formula id="inf69">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> by running the policy <inline-formula id="inf70">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>&#x2003;in the environment</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Calculate the rewards-to-go&#x2003;<inline-formula id="inf71">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Estimate advantages&#x2003;<inline-formula id="inf72">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> using a suitable method based on the current value function <inline-formula id="inf73">
<mml:math id="m78">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Update the policy by maximizing the PPO-Clip objective</p>
</list-item>
<list-item>
<p>
<disp-formula id="equ1">
<mml:math display="block" id="m79">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:mo>&#x2062;</mml:mo>
<mml:munder>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mo>&#x2062;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">D</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">D</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mi>min</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mfenced close=")" open="(" separators="none">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mfenced close=")" open="(" separators="none">
<mml:mi>&#x3b8;</mml:mi>
</mml:mfenced>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo form="prefix" movablelimits="false">clip</mml:mo>
<mml:mfenced close=")" open="(" separators="none">
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>&#x2062;</mml:mo>
<mml:mfenced close=")" open="(" separators="none">
<mml:mi>&#x3b8;</mml:mi>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2062;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>A</mml:mi>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>  </list-item>
<list-item>
<p>&#x2003;where&#x2003;<inline-formula id="inf74">
<mml:math id="m80">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>&#x2003;&#x2003;Update the value function by minimizing the mean-squared error:<disp-formula id="equ2">
<mml:math id="m81">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:mspace width="0.2em"/>
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>V</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
<sec id="s3-2">
<title>3.2 Cooperative Kalman Filter</title>
<p>Cooperative Kalman Filter (CKF) is a collaborative state estimation scheme first developed in (<xref ref-type="bibr" rid="B45">Zhang and Leonard, 2010</xref>), then used in later studies (<xref ref-type="bibr" rid="B40">Wu and Zhang, 2012</xref>; <xref ref-type="bibr" rid="B43">You et al., 2016</xref>; <xref ref-type="bibr" rid="B39">Wu et al., 2020</xref>; <xref ref-type="bibr" rid="B44">You et al., 2022</xref>), by combining live sensor data collected by the network of multiple mobile robots to collaboratively improve the accuracy of the state estimation process. In particular, when applied to the state estimation in dynamic fields, the authors incorporated the dynamics of the mobile robot formation and the diffusion equation into the formulation of the state equation of the CKF. This integration facilitates reliable and accurate state estimation, taking into account how changes in diffusion fields and the formation trajectory over time affect sensor data measurements. More specifically, the state vector <inline-formula id="inf75">
<mml:math id="m82">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> at each time step <inline-formula id="inf76">
<mml:math id="m83">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is defined as: <disp-formula id="equ3">
<mml:math id="m84">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula> where <inline-formula id="inf78">
<mml:math id="m85">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf79">
<mml:math id="m86">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> denote the field concentration values at location <inline-formula id="inf80">
<mml:math id="m87">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> at two consecutive time steps <inline-formula id="inf81">
<mml:math id="m88">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf82">
<mml:math id="m89">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, respectively, and <inline-formula id="inf83">
<mml:math id="m90">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf84">
<mml:math id="m91">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> denote the field gradient at location <inline-formula id="inf85">
<mml:math id="m92">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> at two consecutive time steps <inline-formula id="inf86">
<mml:math id="m93">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf87">
<mml:math id="m94">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, respectively. Given that the mobile robots maintain a symmetrical formation while traversing the environment, CKF estimates the state vector <inline-formula id="inf88">
<mml:math id="m95">
<mml:mrow>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> along the trajectory of the formation center. Note that since the field is spatial-temporal varying, the field concentration values and gradients are different at time steps <inline-formula id="inf89">
<mml:math id="m96">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf90">
<mml:math id="m97">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> even at the same location <inline-formula id="inf91">
<mml:math id="m98">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. This fact is critical in the construction of the CKF to provide reliable estimates of the state vector. The estimated state vector is subsequently employed to iteratively identify the diffusion coefficients of the field over time, based on the RLS algorithm. The estimated diffusion coefficients are vital for the task of identifying and reconstructing spatial diffusion fields in our paper. To save time and avoid excessive length in this paper, we will not provide the complete derivation of the CKF. For additional details, interested readers may refer to the original papers.</p>
</sec>
</sec>
<sec sec-type="methods" id="s4">
<title>4 Methodology</title>
<p>In this section, we introduce the proposed path-planning algorithm for guiding the mobile sensing robot formation to quickly explore an open field while reliably mapping and reconstructing all detected diffusion sources along its trajectory. The algorithm aims to find a balance between speed and reliability for the dynamic field reconstruction. To achieve this, the solution alternates the robot formation between two operational modes: Map Exploration and Source Mapping. In Map Exploration, the robots systematically advance toward unexplored regions to detect new diffusion fields. Upon detecting a new diffusion field, the system transitions to Source Mapping, where the formation converges on the field&#x2019;s center to achieve a stationary state, necessary for estimating advection coefficients.</p>
<p>Throughout both modes, the robots continuously collect data, using the CKF for real-time concentration and gradient estimation and the RLS algorithm for identifying diffusion coefficients. In the discrete simulation environment, concentration estimates are interpolated across the formation&#x2019;s view-scope. Mode transitions are based on the formation&#x2019;s state and the concentration estimates at its center. <xref ref-type="fig" rid="F2">Figure 2</xref> provides an overview of all major components in our algorithm and their interaction within the two operation modes. In the following sections, we will provide details of the algorithms developed for the two modes.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Flow chart showing the key components of our algorithm and two operation modes of the robot formation.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g002.tif"/>
</fig>
<sec id="s4-1">
<title>4.1 Source mapping mode</title>
<p>As discussed in the high-level overview, the goal of the formation in Source Mapping mode is to move toward the source of a diffusion field and facilitate diffusion field reconstruction by estimating advection and diffusion parameters. For this purpose, we train a PPO model that takes the field information vector state to predict the optimal action. In this section, we describe the setup of our training environment and the architecture of our PPO model.</p>
<p>We define the observation input state <inline-formula id="inf92">
<mml:math id="m99">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for our PPO model as follows:<disp-formula id="e6">
<mml:math id="m100">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>&#x2207;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi>&#x2207;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>where <inline-formula id="inf93">
<mml:math id="m101">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf94">
<mml:math id="m102">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> represent the estimated concentration gradients at the formation center at time step <inline-formula id="inf95">
<mml:math id="m103">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, in the <inline-formula id="inf96">
<mml:math id="m104">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf97">
<mml:math id="m105">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> directions, respectively. As discussed in the previous section, we rely on the CKF to provide estimates of the field concentration and gradients at the formation center, which form the complete input state of our model. By incorporating concentration gradients in the observation state, we provide the model with the direction of the largest concentration value change at the current location, which can be useful for heading toward the source center. Additionally, our definition of the state vector <inline-formula id="inf98">
<mml:math id="m106">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="disp-formula" rid="e6">Equation 6</xref> limits the model to learning state-action values solely based on the characteristics of the field. With the formation controller, the mobile sensor robots can maintain a constant desired formation while traversing the environment; thus; we only plan the path for the formation center instead of individual robots.</p>
<p>For every time step, our robot formation can move to any adjacent cells (including diagonal) or stay at the current location. Thus, we can define the action space <inline-formula id="inf99">
<mml:math id="m107">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> consisting of nine actions as follows:<disp-formula id="e7">
<mml:math display="block" id="m108">
<mml:mrow>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2061;</mml:mo>
<mml:mfenced close=")" open="(" separators="none">
<mml:mi>k</mml:mi>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced close="}" open="{" separators="none">
<mml:mrow>
<mml:mspace width=".2em"/>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>up</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mrow>
<mml:mtext>,</mml:mtext>
<mml:mspace width=".2em"/>
</mml:mrow>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>down</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mrow>
<mml:mtext>,</mml:mtext>
<mml:mspace width=".2em"/>
</mml:mrow>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>left</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mrow>
<mml:mtext>,</mml:mtext>
<mml:mspace width=".2em"/>
</mml:mrow>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>right</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mrow>
<mml:mtext>,</mml:mtext>
<mml:mspace width=".2em"/>
</mml:mrow>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>up</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>left</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mrow>
<mml:mtext>,</mml:mtext>
<mml:mspace width=".2em"/>
</mml:mrow>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>up</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>right</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mrow>
<mml:mtext>,</mml:mtext>
<mml:mspace width=".2em"/>
</mml:mrow>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>down</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>left</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mrow>
<mml:mtext>,</mml:mtext>
<mml:mspace width=".2em"/>
</mml:mrow>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>down</mml:mtext>
<mml:mo>-</mml:mo>
<mml:mtext>right</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mrow>
<mml:mtext>,</mml:mtext>
<mml:mspace width=".2em"/>
</mml:mrow>
<mml:mi>&#x201c;</mml:mi>
<mml:mtext>stay</mml:mtext>
<mml:mi>&#x201d;</mml:mi>
<mml:mspace width=".2em"/>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>
</p>
<p>Since our goal is to train a PPO model that can guide the formation toward the center of a diffusion field and maintain stationary state as long as possible, it is crucial to develop a reward function that incentivizes this behavior. For this reason, we model the reward function based on the concentration values inside the formation view-scope as follows:<disp-formula id="e8">
<mml:math id="m109">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="normal">&#x393;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf100">
<mml:math id="m110">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a rescale constant. As regions with high concentration values play a significant role in field reconstruction, a reward proportional to the total concentration values within the view-scope as shown in <xref ref-type="disp-formula" rid="e8">Equation 8</xref> motivates the model to learn to navigate towards areas with higher concentration values. This, in turn, greatly reduces the error in reconstruction and prioritizes information-rich trajectory. We chose PPO as our model due to its long-standing role as a crucial component in various state-of-the-art solutions in reinforcement learning. Furthermore, PPO can be used for environments with discrete action space, as in our case. <xref ref-type="fig" rid="F3">Figure 3</xref> shows the architecture of our PPO model. Since PPO is a type of actor-critic algorithm, it has two neural networks - the actor network and critic network. In our case, we employed the same architecture for both. This network consists of three dense layers of size 128 each, an input layer of size 3 and an output layer of size 9. We added random dropout layers and regularization between the hidden layers, with ReLu as the non-linear activation function (<xref ref-type="bibr" rid="B1">Agarap, 2018</xref>). The network is trained using Adams optimizer (<xref ref-type="bibr" rid="B15">Kingma and Ba, 2015</xref>).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The PPO neural network architecture used in the source mapping mode.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g003.tif"/>
</fig>
<p>Using the trained PPO model, the robot formation takes actions chosen from the action space in <xref ref-type="disp-formula" rid="e7">Equation 7</xref>, and is guided toward the source of the diffusion field until it reaches the stationary state. This stationary state is achieved when the estimated field concentration reaches a local maximum and the estimated field gradient approaches zero, i.e., <inline-formula id="inf101">
<mml:math id="m111">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf102">
<mml:math id="m112">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf103">
<mml:math id="m113">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2248;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. At this point, the formation moves at the same speed as the advection flow for a designated period of <inline-formula id="inf104">
<mml:math id="m114">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> steps before switching to the Field Exploration mode to search for other diffusion fields in unexplored areas.</p>
</sec>
<sec id="s4-2">
<title>4.2 Field exploration mode</title>
<p>Whenever the robot formation reaches a stationary state, indicating the detection of the source of a diffusion field, the robot formation switches back to Field Exploration mode and moves toward unexplored area in the map to look for the remaining diffusion fields. During this phase, it is essential for the robot formation to come up with a new destination that is away from the already explored locations to avoid revisiting the same source, but also not too far to cause the formation to go back and forth when scanning the whole map. In short, our main objective is to generate a path that allows our robot formation to scan the whole field as quickly as possible without leaving any diffusion field undetected. Lawn mowing is a good example of generating a deterministic trajectory for scanning an unknown map. However, since our mobile sensing robots are often deployed in time-critical missions. It is necessary to opt for a more aggressive exploration strategy that allows the formation to discover all sources as quickly as possible. In our algorithm, we partition the unvisited cells in the entire map into multiple clusters using the K-Means clustering algorithm (<xref ref-type="bibr" rid="B20">Na et al., 2010</xref>). The selection of <inline-formula id="inf105">
<mml:math id="m115">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> directly affects how aggressive or cautious the scanning behavior of our robot formation will be. The initial value of <inline-formula id="inf106">
<mml:math id="m116">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is selected based on the initial estimate of the minimum size of a diffusion field. In this work, we use the term &#x201c;size&#x201d; to refer to the bounded area of a diffusion field where the concentration value exceeds a certain threshold. For different scenarios (such as wildfires and gas leaks), we are often able to come up with a rough estimation of the average size of a diffusion field. Let us denote this as <inline-formula id="inf107">
<mml:math id="m117">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">field</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the global map size as <inline-formula id="inf108">
<mml:math id="m118">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">map</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Then, <inline-formula id="inf109">
<mml:math id="m119">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is estimated as:<disp-formula id="e9">
<mml:math id="m120">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>,</mml:mo>
<mml:mfenced open="&#x230a;" close="&#x230b;">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">map</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">field</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>After partitioning the map, the robot formation selects the centroid of the nearest cluster as the new destination and moves toward that destination to explore the field. It continues to visit the centroids of other clusters, prioritizing nearby clusters, as long as it is in the Field Exploration mode. The formation switches to Source Mapping mode when the estimated field concentration value exceeds a chosen threshold, i.e., <inline-formula id="inf110">
<mml:math id="m121">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3e;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. At all times, the formation maintains a record of visited clusters and diffusion fields, effectively creating a mask that distinguishes between explored and unexplored regions.</p>
<p>Whenever a switch occurs from Field Exploration mode to Source Mapping mode and the formation reaches the stationary state in the Source Mapping mode (indicating a new diffusion field is detected), the robot formation evaluates the field and computes a new estimated <inline-formula id="inf111">
<mml:math id="m122">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> value to repartition the remaining unexplored areas in the map. The &#x201c;size&#x201d; of the newly discovered diffusion field can be roughly estimated based on the circular area with radius <inline-formula id="inf112">
<mml:math id="m123">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">dist</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> extending from the source center to the location where the concentration first surpasses the threshold. This is also where the formation switches from Field Exploration to Source Mapping mode. Denote the size as <inline-formula id="inf113">
<mml:math id="m124">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">new</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represented as:<disp-formula id="e10">
<mml:math id="m125">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">new</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2248;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">dist</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>With the size of the latest detected diffusion field calculated based on <xref ref-type="disp-formula" rid="e10">Equation 10</xref>, we can update the estimated average size of the diffusion field <inline-formula id="inf114">
<mml:math id="m126">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">field</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as follows:<disp-formula id="e11">
<mml:math id="m127">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">field</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">field</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">new</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>where <inline-formula id="inf115">
<mml:math id="m128">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the weighting factor that determines the influence of the newly discovered field on the updated estimate <inline-formula id="inf116">
<mml:math id="m129">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mo>&#x303;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="italic">field</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. This approach allows for increasing accurate field size estimation with new discovery. Note that <inline-formula id="inf117">
<mml:math id="m130">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> will only decrease or maintain unchanged with each new source found. When the size of an unexplored area in the map falls below a certain threshold, we consider the whole map has been adequately covered. At this point, The robot formation can either be set to idle mode or be directed to transition to a new map (potentially a neighboring global field) to initiate its operations from a different starting location. The exploration strategy can be summarized in <xref ref-type="statement" rid="Algorithm_2">Algorithm 2</xref>.</p>
<p>
<statement content-type="algorithm" id="Algorithm_2">
<label>Algorithm 2</label>
<p>K-Means Clustering Based Exploration Mode.<list list-type="simple">
<list-item>
<p>1:&#x2003;<bold>Input:</bold> Unexplored region <inline-formula id="inf118">
<mml:math id="m131">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> as an <inline-formula id="inf119">
<mml:math id="m132">
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> grid map.</p>
</list-item>
<list-item>
<p>2:&#x2003;<bold>Initialize:</bold>
</p>
</list-item>
<list-item>
<p>3:&#x2003;Compute initial estimate <inline-formula id="inf120">
<mml:math id="m133">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> based on diffusion field radius using <xref ref-type="disp-formula" rid="e9">Equation 9</xref>.</p>
</list-item>
<list-item>
<p>4:&#x2003;Apply K-Means Clustering with <inline-formula id="inf121">
<mml:math id="m134">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to partition <inline-formula id="inf122">
<mml:math id="m135">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> into <inline-formula id="inf123">
<mml:math id="m136">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> clusters.</p>
</list-item>
<list-item>
<p>5:&#x2003;Let <inline-formula id="inf124">
<mml:math id="m137">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>centroids</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> be the set of centroids for these clusters.</p>
</list-item>
<list-item>
<p>6:&#x2003;Set concentration threshold <inline-formula id="inf125">
<mml:math id="m138">
<mml:mrow>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> for mode switching.</p>
</list-item>
<list-item>
<p>7:&#x2003;Set target destination <inline-formula id="inf126">
<mml:math id="m139">
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
<mml:mi mathvariant="bold">a</mml:mi>
<mml:mi mathvariant="bold">r</mml:mi>
<mml:mi mathvariant="bold">g</mml:mi>
<mml:mi mathvariant="bold">e</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>8:&#x2003;Let DONE <inline-formula id="inf127">
<mml:math id="m140">
<mml:mrow>
<mml:mo>&#x2190;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> False.</p>
</list-item>
<list-item>
<p>9:&#x2003;<bold>while</bold> <inline-formula id="inf128">
<mml:math id="m141">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3c;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <bold>do</bold>
</p>
</list-item>
<list-item>
<p>10:&#x2003;&#x2003;<bold>if</bold> formation center <inline-formula id="inf129">
<mml:math id="m142">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold">T</mml:mi>
<mml:mi mathvariant="bold">a</mml:mi>
<mml:mi mathvariant="bold">r</mml:mi>
<mml:mi mathvariant="bold">g</mml:mi>
<mml:mi mathvariant="bold">e</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <bold>then</bold>
</p>
</list-item>
<list-item>
<p>11:&#x2003;&#x2003;&#x2003;<bold>if</bold> <inline-formula id="inf130">
<mml:math id="m143">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>centroids</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf131">
<mml:math id="m144">
<mml:mrow>
<mml:mo>&#x2260;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf132">
<mml:math id="m145">
<mml:mrow>
<mml:mi>&#x2205;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> <bold>then</bold>
</p>
</list-item>
<list-item>
<p>12:&#x2003;&#x2003;&#x2003;&#x2003;Set <inline-formula id="inf133">
<mml:math id="m146">
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
<mml:mi mathvariant="bold">a</mml:mi>
<mml:mi mathvariant="bold">r</mml:mi>
<mml:mi mathvariant="bold">g</mml:mi>
<mml:mi mathvariant="bold">e</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as the nearest centroid <inline-formula id="inf134">
<mml:math id="m147">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>nearest</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> in <inline-formula id="inf135">
<mml:math id="m148">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>centroids</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>13:&#x2003;&#x2003;&#x2003;&#x2003;Remove <inline-formula id="inf136">
<mml:math id="m149">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>nearest</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> from <inline-formula id="inf137">
<mml:math id="m150">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>centroids</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</list-item>
<list-item>
<p>14:&#x2003;&#x2003;&#x2003;<bold>else</bold>
</p>
</list-item>
<list-item>
<p>15:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;DONE <inline-formula id="inf138">
<mml:math id="m151">
<mml:mrow>
<mml:mo>&#x2190;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> True</p>
</list-item>
<list-item>
<p>16:&#x2003;&#x2003;&#x2003;&#x2003;&#x2003;<bold>break</bold>
</p>
</list-item>
<list-item>
<p>17:&#x2003;&#x2003;Move formation towards <inline-formula id="inf139">
<mml:math id="m152">
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
<mml:mi mathvariant="bold">a</mml:mi>
<mml:mi mathvariant="bold">r</mml:mi>
<mml:mi mathvariant="bold">g</mml:mi>
<mml:mi mathvariant="bold">e</mml:mi>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> using A&#x2a; path-finding algorithm.</p>
</list-item>
<list-item>
<p>18:&#x2003;<bold>if</bold> DONE <bold>then</bold>
</p>
</list-item>
<list-item>
<p>19:&#x2003;&#x2003;Transition to Source Mapping Mode.</p>
</list-item>
<list-item>
<p>20:&#x2003;<bold>else</bold>
</p>
</list-item>
<list-item>
<p>21:&#x2003;&#x2003;Move to new region <inline-formula id="inf140">
<mml:math id="m153">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and restart exploration.</p>
</list-item>
</list>
</p>
</statement>
</p>
</sec>
<sec id="s4-3">
<title>4.3 Parameter estimation and field reconstruction</title>
<p>The field reconstruction begins when the formation detects the source of a diffusion field within the global domain. Given that our environment is modeled as a 2D grid, we discretize the diffusion <xref ref-type="disp-formula" rid="e1">Equation 1</xref> to enable field reconstruction. Assuming the domain of interest <inline-formula id="inf141">
<mml:math id="m154">
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a9;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is divided into square cells of size <inline-formula id="inf142">
<mml:math id="m155">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, as illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>, where a <inline-formula id="inf143">
<mml:math id="m156">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> grid is demonstrated. Here, <inline-formula id="inf144">
<mml:math id="m157">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the neighboring cells of grid cell <inline-formula id="inf145">
<mml:math id="m158">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Let <inline-formula id="inf146">
<mml:math id="m159">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf147">
<mml:math id="m160">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denote the concentration values at grid cells <inline-formula id="inf148">
<mml:math id="m161">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf149">
<mml:math id="m162">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> at discrete time step <inline-formula id="inf150">
<mml:math id="m163">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Using the finite difference method, the discretized advection-diffusion equation can be expressed as:<disp-formula id="e12">
<mml:math id="m164">
<mml:mtable class="align" columnalign="left">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mfrac>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mfenced open="[" close="">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mspace width="1em"/>
<mml:mfenced open="" close="]">
<mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:mspace width="1em"/>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
<label>(12)</label>
</disp-formula>where <inline-formula id="inf151">
<mml:math id="m165">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the sampling interval and <inline-formula id="inf152">
<mml:math id="m166">
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the approximation error. With the symmetric property, <xref ref-type="disp-formula" rid="e12">Equation 12</xref> can be further simplified to:<disp-formula id="e13">
<mml:math id="m167">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mi>&#x2207;</mml:mi>
<mml:mi>z</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>e</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>
</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>A <inline-formula id="inf153">
<mml:math id="m168">
<mml:mrow>
<mml:mn>3</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> section of the discretized advection-diffusion field.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g004.tif"/>
</fig>
<p>To reconstruct a diffusion field using the measurements taken by the robot formation with <xref ref-type="disp-formula" rid="e13">Equation 13</xref>, we need the estimated field concentration values <inline-formula id="inf154">
<mml:math id="m169">
<mml:mrow>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> within the view-scope of the robot formation at each time step <inline-formula id="inf155">
<mml:math id="m170">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, as well as the estimated diffusion coefficient <inline-formula id="inf156">
<mml:math id="m171">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> and the advection vector <inline-formula id="inf157">
<mml:math id="m172">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>. The former can be obtained through interpolation at each time step using the measurements taken by the robots. As mentioned previously, we employ the strategy developed in (<xref ref-type="bibr" rid="B39">Wu et al., 2020</xref>) to identify the diffusion coefficient <inline-formula id="inf158">
<mml:math id="m173">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. In many scenarios, the advection coefficient <inline-formula id="inf159">
<mml:math id="m174">
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is assumed to be a known constant. When the advection coefficient is unknown, we estimate it when the robot formation reaches a stationary state, where the robot&#x2019;s velocity matches that of the advection flow. Let <inline-formula id="inf160">
<mml:math id="m175">
<mml:mrow>
<mml:mi mathvariant="bold">v</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf161">
<mml:math id="m176">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> represent the time step when a stationary state is detected. Assuming the formation stays at the stationary state for <inline-formula id="inf162">
<mml:math id="m177">
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> steps, <inline-formula id="inf163">
<mml:math id="m178">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> can be estimated as<disp-formula id="e14">
<mml:math id="m179">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="right"/>
<mml:mtd columnalign="left">
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">&#x302;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>.</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
</p>
<p>With these values determined, the field values across the diffusion field can be propagated through <xref ref-type="disp-formula" rid="e13">Equation 13</xref>. This approach enables field reconstruction using only the sparse measurements gathered along the robots&#x2019; paths.</p>
</sec>
</sec>
<sec id="s5">
<title>5 Simulation results</title>
<p>In this section, we provide a comprehensive analysis of the proposed multi-robot field reconstruction strategy, which encompasses source mapping and field exploration modes in simulations. We begin by outlining the implementation details, followed by a discussion on the PPO training specifics. Finally, we present the results derived from these simulations.</p>
<sec id="s5-1">
<title>5.1 Simulation environment</title>
<p>To assess the overall solution, we developed a low-fidelity simulation environment within a discrete space. This environment is structured as a <inline-formula id="inf164">
<mml:math id="m180">
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> square grid, incorporating between one to four non-overlapping spatial diffusion fields of varying sizes and configurations, as depicted in <xref ref-type="fig" rid="F5">Figure 5</xref>. We generated a total of 15 different environments, with increasing level of complexity, to investigate and assess the model&#x2019;s efficiency in mapping unknown environments. The color of each cell in the grid denotes the concentration value of the diffusion field. Each diffusion field possesses distinct and independent advection and diffusion coefficients. <xref ref-type="fig" rid="F6">Figure 6</xref> shows the evolution of a sample diffusion field over time. The environment was simulated for up to 400 time steps with the discrete interval <inline-formula id="inf165">
<mml:math id="m181">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf166">
<mml:math id="m182">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf167">
<mml:math id="m183">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. <xref ref-type="table" rid="T1">Table 1</xref> lists the configurations of the 15 diffusion field environments with advection terms <inline-formula id="inf168">
<mml:math id="m184">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and diffusion term <inline-formula id="inf169">
<mml:math id="m185">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The center of the field is denoted as <inline-formula id="inf170">
<mml:math id="m186">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Note that <inline-formula id="inf171">
<mml:math id="m187">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is only used internally by the generator as a scale factor to control how large a diffusion field appears on the map.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>A set of 15 simulation environments, each consisting of one to four diffusion fields centered at various locations with distinct diffusion and advection coefficients.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g005.tif"/>
</fig>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Sample of a diffusion field environment over different time steps.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g006.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Configurations of the 15 diffusion field environments with advection terms <inline-formula id="inf172">
<mml:math id="m188">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and diffusion term <inline-formula id="inf173">
<mml:math id="m189">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. The center of the field is denoted as <inline-formula id="inf174">
<mml:math id="m190">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Note that <inline-formula id="inf175">
<mml:math id="m191">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is only used internally by the generator as a scale factor to control how large a diffusion field appears on the map.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Source</th>
<th align="center">pos (x,y)</th>
<th align="center">Size</th>
<th align="center">
<inline-formula id="inf176">
<mml:math id="m192">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">
<inline-formula id="inf177">
<mml:math id="m193">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>v</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center">
<inline-formula id="inf178">
<mml:math id="m194">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">1-src-1</td>
<td align="center">(50, 50)</td>
<td align="center">80</td>
<td align="center">0.73</td>
<td align="center">&#x2212;0.44</td>
<td align="center">0.76</td>
</tr>
<tr>
<td align="center">1-src-2</td>
<td align="center">(30, 70)</td>
<td align="center">60</td>
<td align="center">0.14</td>
<td align="center">&#x2212;0.65</td>
<td align="center">1.02</td>
</tr>
<tr>
<td align="center">1-src-3</td>
<td align="center">(90, 10)</td>
<td align="center">160</td>
<td align="center">0.74</td>
<td align="center">0.09</td>
<td align="center">1.24</td>
</tr>
<tr>
<td rowspan="2" align="center">2-src-1</td>
<td align="center">(20, 20)</td>
<td align="center">100</td>
<td align="center">0.62</td>
<td align="center">&#x2212;0.01</td>
<td align="center">1.07</td>
</tr>
<tr>
<td align="center">(80, 80)</td>
<td align="center">100</td>
<td align="center">&#x2212;0.04</td>
<td align="center">&#x2212;0.14</td>
<td align="center">0.96</td>
</tr>
<tr>
<td rowspan="2" align="center">2-src-2</td>
<td align="center">(50, 20)</td>
<td align="center">60</td>
<td align="center">&#x2212;0.22</td>
<td align="center">&#x2212;0.47</td>
<td align="center">1.04</td>
</tr>
<tr>
<td align="center">(50, 80)</td>
<td align="center">60</td>
<td align="center">&#x2212;0.59</td>
<td align="center">&#x2212;0.45</td>
<td align="center">1.08</td>
</tr>
<tr>
<td rowspan="2" align="center">2-src-3</td>
<td align="center">(20, 20)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.71</td>
<td align="center">0.09</td>
<td align="center">1.17</td>
</tr>
<tr>
<td align="center">(50, 80)</td>
<td align="center">80</td>
<td align="center">0.56</td>
<td align="center">&#x2212;0.67</td>
<td align="center">0.78</td>
</tr>
<tr>
<td rowspan="2" align="center">2-src-4</td>
<td align="center">(20, 80)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.20</td>
<td align="center">0.37</td>
<td align="center">0.75</td>
</tr>
<tr>
<td align="center">(90, 10)</td>
<td align="center">120</td>
<td align="center">0.71</td>
<td align="center">0.78</td>
<td align="center">0.99</td>
</tr>
<tr>
<td rowspan="2" align="center">2-src-5</td>
<td align="center">(40, 80)</td>
<td align="center">60</td>
<td align="center">0.10</td>
<td align="center">0.38</td>
<td align="center">1.15</td>
</tr>
<tr>
<td align="center">(80, 30)</td>
<td align="center">60</td>
<td align="center">&#x2212;0.09</td>
<td align="center">0.25</td>
<td align="center">1.07</td>
</tr>
<tr>
<td rowspan="3" align="center">3-src-1</td>
<td align="center">(10, 10)</td>
<td align="center">100</td>
<td align="center">0.54</td>
<td align="center">0.19</td>
<td align="center">1.21</td>
</tr>
<tr>
<td align="center">(50, 90)</td>
<td align="center">100</td>
<td align="center">0.03</td>
<td align="center">0.58</td>
<td align="center">0.94</td>
</tr>
<tr>
<td align="center">(90, 10)</td>
<td align="center">100</td>
<td align="center">&#x2212;0.64</td>
<td align="center">0.67</td>
<td align="center">0.79</td>
</tr>
<tr>
<td rowspan="3" align="center">3-src-2</td>
<td align="center">(40, 40)</td>
<td align="center">80</td>
<td align="center">0.39</td>
<td align="center">&#x2212;0.50</td>
<td align="center">0.81</td>
</tr>
<tr>
<td align="center">(80, 80)</td>
<td align="center">80</td>
<td align="center">0.60</td>
<td align="center">0.25</td>
<td align="center">0.93</td>
</tr>
<tr>
<td align="center">(90, 10)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.12</td>
<td align="center">0.11</td>
<td align="center">1.18</td>
</tr>
<tr>
<td rowspan="3" align="center">3-src-3</td>
<td align="center">(50, 50)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.37</td>
<td align="center">0.24</td>
<td align="center">1.13</td>
</tr>
<tr>
<td align="center">(10, 90)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.32</td>
<td align="center">&#x2212;0.75</td>
<td align="center">1.21</td>
</tr>
<tr>
<td align="center">(90, 10)</td>
<td align="center">80</td>
<td align="center">0.76</td>
<td align="center">0.71</td>
<td align="center">1.02</td>
</tr>
<tr>
<td rowspan="3" align="center">3-src-4</td>
<td align="center">(50, 5)</td>
<td align="center">120</td>
<td align="center">&#x2212;0.49</td>
<td align="center">0.17</td>
<td align="center">0.91</td>
</tr>
<tr>
<td align="center">(30, 80)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.50</td>
<td align="center">0.56</td>
<td align="center">0.84</td>
</tr>
<tr>
<td align="center">(80, 70)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.44</td>
<td align="center">0.42</td>
<td align="center">0.98</td>
</tr>
<tr>
<td rowspan="3" align="center">3-src-5</td>
<td align="center">(20, 20)</td>
<td align="center">80</td>
<td align="center">0.58</td>
<td align="center">0.45</td>
<td align="center">0.97</td>
</tr>
<tr>
<td align="center">(80, 80)</td>
<td align="center">80</td>
<td align="center">0.66</td>
<td align="center">&#x2212;0.18</td>
<td align="center">0.99</td>
</tr>
<tr>
<td align="center">(30, 80)</td>
<td align="center">80</td>
<td align="center">0.73</td>
<td align="center">&#x2212;0.23</td>
<td align="center">0.99</td>
</tr>
<tr>
<td rowspan="4" align="center">4-src-1</td>
<td align="center">(10, 10)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.08</td>
<td align="center">0.22</td>
<td align="center">1.14</td>
</tr>
<tr>
<td align="center">(90, 10)</td>
<td align="center">80</td>
<td align="center">0.37</td>
<td align="center">0.35</td>
<td align="center">0.77</td>
</tr>
<tr>
<td align="center">(90, 90)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.72</td>
<td align="center">&#x2212;0.38</td>
<td align="center">0.76</td>
</tr>
<tr>
<td align="center">(10, 90)</td>
<td align="center">80</td>
<td align="center">0.37</td>
<td align="center">0.63</td>
<td align="center">0.86</td>
</tr>
<tr>
<td rowspan="4" align="center">4-src-2</td>
<td align="center">(50, 10)</td>
<td align="center">80</td>
<td align="center">0.72</td>
<td align="center">&#x2212;0.73</td>
<td align="center">1.05</td>
</tr>
<tr>
<td align="center">(50, 90)</td>
<td align="center">80</td>
<td align="center">&#x2212;0.75</td>
<td align="center">0.74</td>
<td align="center">1.06</td>
</tr>
<tr>
<td align="center">(5, 50)</td>
<td align="center">100</td>
<td align="center">0.35</td>
<td align="center">&#x2212;0.05</td>
<td align="center">1.03</td>
</tr>
<tr>
<td align="center">(95, 50)</td>
<td align="center">100</td>
<td align="center">0.12</td>
<td align="center">0.73</td>
<td align="center">1.25</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Since the map is discretized in our approach, selecting an appropriate grid cell size also plays a role in both the accuracy of data collection and computational efficiency. Each grid cell should be small enough to capture meaningful concentration gradients but large enough to reduce computational demands. Ideally, the grid cell size should reflect both the overall map dimensions and the characteristics of the environment being monitored. For example, when studying gas leaks, where subtle concentration changes are significant, a finer grid may be required. On the other hand, wildfire propagation fields, which tend to cover larger areas, can accommodate slightly larger cells. Adjusting the grid cell size based on the specific characteristics of the environment allows us to find the right balance between resolution and computational cost, facilitating effective and efficient exploration and reconstruction.</p>
</sec>
<sec id="s5-2">
<title>5.2 PPO training</title>
<p>For the training of the PPO model, we follow a curriculum learning approach (<xref ref-type="bibr" rid="B37">Wang et al., 2023</xref>; <xref ref-type="bibr" rid="B38">Wang et al., 2022</xref>) that involves gradually increasing the complexity of the training environment. We created a <inline-formula id="inf179">
<mml:math id="m195">
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> training environment featuring a single diffusion field at the center of the map <inline-formula id="inf180">
<mml:math id="m196">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>50,50</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> with a constant diffusion coefficient <inline-formula id="inf181">
<mml:math id="m197">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The spread of this diffusion field at the beginning of an episode is drawn randomly based on the parameter <inline-formula id="inf182">
<mml:math id="m198">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, which controls how large the area with non-zero concentration is due to the presence of the diffusion field. This environment has two variants: a &#x201c;static&#x201d; environment with the advection term set to zero, and a &#x201c;dynamic&#x201d; environment with a fixed non-zero advection term.</p>
<p>We simulate a group of four mobile robots in a symmetric formation as shown in <xref ref-type="fig" rid="F1">Figure 1</xref> to move in the environments, with the formation controller running to maintain the desired formation. With the CKF providing the state <inline-formula id="inf183">
<mml:math id="m199">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> which includes the estimated field value <inline-formula id="inf184">
<mml:math id="m200">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and gradient along the trajectory of the formation center <inline-formula id="inf185">
<mml:math id="m201">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf186">
<mml:math id="m202">
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, the PPO model underwent training on static maps first before advancing to training on dynamic maps. After training, our PPO model outputs the policy to direct the robot formation to move toward the source of a diffusion field and maintain the stationary state, which is required for the estimation of advection coefficients.</p>
<p>We provide the training results in <xref ref-type="fig" rid="F7">Figure 7</xref>. The PPO model was initially trained for 400,000 time steps on the static environment where advection terms are set to zero, as shown in <xref ref-type="fig" rid="F7">Figure 7A</xref>. After that, we proceeded to train the PPO model on the dynamic environment for an additional 2,000,000 time steps, as shown in <xref ref-type="fig" rid="F7">Figure 7B</xref> The model was trained multiple times with orthogonal random weight initialization using the Stable-Baselines3 framework (<xref ref-type="bibr" rid="B25">Raffin et al., 2021</xref>), and the best-performing model was selected for use in our experiments. In both environments, the formation is initially placed in a low-concentration region at the start of each episode to avoid starting too close to the source. Additionally, in our dynamic environment, the source is assigned random, non-zero advection and diffusion coefficients, causing it to move in a different direction in each episode. This setup encourages the model to adapt to various scenarios but also introduces some fluctuations in performance early in training, as the formation may take suboptimal actions initially and struggle to catch up to the moving source. In both training phases, however, the average reward per episode increases steadily, indicating that the model successfully improves its given task over time. Additionally, <xref ref-type="fig" rid="F7">Figure 7C</xref> provide some samples of source-heading operation performed by our PPO model post-training. In these samples, the 4-robot formation, shown as the red square in the map, are tasked with locating the source while maintaining its formation. The yellow region denotes the area with high concentration - where the source is located, the red square represents the robot formation, and the red dot is the starting position of the formation center. The robot formation is directed towards the center of the source, which has maximum concentration, per the PPO&#x2019;s objective of reward maximization. PPO&#x2019;s role in this task is to update the policy per iteration to make an informed decision on where to go next. The simulation results show the PPO algorithm&#x2019;s efficacy in source mapping.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>
<bold>(A)</bold> Stage 1: Pre-training PPO on static fields. <bold>(B)</bold> Stage 2: Training PPO on dynamic fields. <bold>(C)</bold> Samples of training environments and generated trajectories post-training.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g007.tif"/>
</fig>
<p>The PPO algorithm&#x2019;s training process effectively learns a policy that guides the robot formation to explore the field, detect diffusion sources, and assist in reconstruction while adapting to dynamic changes in the environment. The algorithm&#x2019;s capability in handling complex and dynamic environments indicates its potential in real-life scenarios where rapid environmental change happens, and accurate detection of diffusion sources is crucial. This capability is precious in pollution tracking or gas leak detection scenarios, where time-sensitive and precise localization is essential.</p>
</sec>
<sec id="s5-3">
<title>5.3 K-means clustering-based exploration</title>
<p>When the PPO model leads the robot formation to move toward a diffusion source and the formation center reaches the stationary state, the advection coefficient can be estimated based on <xref ref-type="disp-formula" rid="e14">Equation 14</xref>, and the field reconstruction process can start using <xref ref-type="disp-formula" rid="e13">Equation 13</xref>. Along with the field reconstruction process, the robot formation switches to the field exploration mode, where the K-means clustering-based exploration algorithm 2 plays a role.</p>
<p>
<xref ref-type="fig" rid="F8">Figure 8</xref> provides an example of how the K-means clustering-based exploration works in a <inline-formula id="inf187">
<mml:math id="m203">
<mml:mrow>
<mml:mn>100</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>100</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> grid map with two diffusion fields. The formation started in Field Exploration mode with the starting position shown as the red dot. Based on our initial assumption about the average size of diffusion fields, the map is partitioned into six clusters with <inline-formula id="inf188">
<mml:math id="m204">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> as illustrated in <xref ref-type="fig" rid="F8">Figure 8B</xref>. The formation moves toward the centroid of the nearest cluster and continues to other neighboring centroids until an area with a high concentration is detected. When that happens, the formation transitions to Source Mapping mode and attempts to move toward the source center to reconstruct the field. When the new diffusion field is fully reconstructed, the formation switches back to Field Exploration mode, re-calculates a new <inline-formula id="inf189">
<mml:math id="m205">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> as described in <xref ref-type="disp-formula" rid="e9">Equations 9</xref>-<xref ref-type="disp-formula" rid="e11">11</xref> and re-partitions the unexplored area, as shown in <xref ref-type="fig" rid="F8">Figure 8C</xref>. Note that the unexplored area has excluded visited clusters and any detected diffusion fields. This process continues to repeat until the map is fully covered.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Field Re-partitioning Behavior of the Exploration Module using K-Mean Clustering. The red dot is the starting position of the formation center. Partitioning happens at the beginning of an episode and whenever a new source is detected. <bold>(A)</bold> The trajectory of the formation center. <bold>(B)</bold> The initial partition of the field. <bold>(C)</bold> The updated partition after the first source on the upper right corner is detected. <bold>(D)</bold> The updated partition after the second source on the lower left corner is detected.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g008.tif"/>
</fig>
</sec>
<sec id="s5-4">
<title>5.4 Multi-robot source seeking and field exploration results</title>
<p>With the trained PPO model and the K-means clustering-based exploration algorithm, we now ready to implement the overall multi-robot Source Mapping and Field Exploration strategy to reconstruct a dynamic field. <xref ref-type="fig" rid="F9">Figure 9</xref> shows different trajectories of the robot formation obtained from the simulation across multiple spatial-diffusion environments. In this setup, the robot formation started at the same initial position <inline-formula id="inf190">
<mml:math id="m206">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>90,10</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> (top-right corner, shown as a red marker). The red square displays the final position of the robot formation at the end of the episode <inline-formula id="inf191">
<mml:math id="m207">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>300</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Despite variations in spatial-diffusion field distributions and characteristics, we can see that the robot formation managed to detect all sources in the map while efficiently covering the entire map before the episode ends.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Generated trajectories obtained from simulations across different spatial-diffusion environments. The robot formation center is initially located at <inline-formula id="inf192">
<mml:math id="m208">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>90,10</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> - shown as the red marker, and the red square shows the position of the formation the end of the episode.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g009.tif"/>
</fig>
<p>In <xref ref-type="fig" rid="F10">Figure 10</xref>, we compare the trajectory and mapping errors of our solution against other alternative path-planning algorithms, including Random-Walking (shown in black) and Lawn-Mowing (shown in green). With the Random-Walking strategy, the formation takes a random action in every time-step, resulting in an arbitrary trajectory. On the other hand, with Lawn-Mowing strategy, the formation attempts to scan the field row-by-row until the entire field is fully covered. Compared to these approaches, the trajectory generated by our solution is faster at detecting and mapping diffusion fields, resulting in a much lower mapping error. In this map, our K-Mean-based approach is the only one that detects all diffusion fields before the episode ends <inline-formula id="inf193">
<mml:math id="m209">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>300</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Analysis of the generated trajectory and mapping errors of our solution in comparison with the Lawn Mowing and Random Walking approaches.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g010.tif"/>
</fig>
</sec>
</sec>
<sec id="s6">
<title>6 Experimental results</title>
<p>To validate our solution in a real-world setting, we developed a high-fidelity testing environment in our lab. Our setup includes four mobile robots operating in a 12 &#xd7; 12 square foot open field that simulates an actual advection-diffusion environment. <xref ref-type="fig" rid="F11">Figure 11</xref> shows our laboratory setup of the mobile sensor network consisting of 4 mobile robots with motion tracking enabled, allowing for accurate collection of real-time trajectory data. The robots are two-wheel differential drive and ROS-based, running on the Jetson Nano (<xref ref-type="bibr" rid="B8">Developer Nvidia, 2024</xref>) computing platform. Each robot is equipped with a 2D Lidar scanner [YDLidar-G4 (<xref ref-type="bibr" rid="B41">YDLIDAR-G4-Datasheet, 2024</xref>)], a speed encoder, and an IMU (BNO-055 (<xref ref-type="bibr" rid="B12">Industries, 2024</xref>)). To enable low-latency sensor fusion, a Teensy 4.0 (<xref ref-type="bibr" rid="B22">PJRC, 2024</xref>) collects and preprocesses sensor readings from the speed encoder and IMU before streaming the results to the main board via rosserial. Lidar is installed to enable basic obstacle avoidance behaviors, allowing the formation to adapt to various scenarios when navigating in outdoor environments.</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Mobile robot formation setup for real-world testing and evaluation.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g011.tif"/>
</fig>
<p>For localization, we rely on an indoor motion capture system to provide absolute positional tracking, analogous to GPS in outdoor scenarios. An Extended Kalman Filter (<xref ref-type="bibr" rid="B28">Ribeiro, 2004</xref>) fuses data from both the IMU and motion capture system to improve real-time position estimation of the robots. Our software stack uses ROS Noetic (<xref ref-type="bibr" rid="B24">Quigley et al., 2009</xref>) and its ecosystem to facilitate sensor fusion for localization and obstacle avoidance, as well as to simulate and visualize the behaviors of the advection-diffusion field. In our stack, each robot has its own action server (based on ROS Action), which is responsible for moving the robot to a target destination. A master node running on a stand-alone computer is responsible for broadcasting the concentration values of the simulated field as well as performing formation control during the experiment.</p>
<p>
<xref ref-type="fig" rid="F12">Figure 12</xref> presents various views of the simulated spatial-temporal diffusion field in our laboratory as well as an example of the trajectory generated by our robot formation. Given the difficulties of installing physical diffusion field sources indoors, we utilized computational models to simulate the environment. The simulated field is projected onto the floor in real-time footage captured by side and top-down cameras. Sensor measurements are generated based on the robots&#x2019; locations, which are tracked using the motion capture system.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>
<bold>(a)</bold> Lab experiment setup with a projected dynamic field and four mobile robots. <bold>(b)</bold> Snapshots of the trajectories of the robot formation at three different time steps in an experiment. The red dashed lines represent the trajectories.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g012.tif"/>
</fig>
<p>To validate our solution in real-world settings, we selected two spatial-diffusion environments from our list and ran simulation experiments using actual mobile robots. While the spatial-temporal diffusion field is generated by computer simulation, the mobile sensor network is still designed to function exactly like how they should behave in the real-world. This involves having individual mobile robots take raw measurements and combine the results to estimate the concentration and gradients at the formation center, using CKF. <xref ref-type="fig" rid="F13">Figure 13</xref> provides a summary of our experimental results. The first column &#x201c;Environment Field End State&#x201d; shows the final states of our spatial diffusion environments and the trajectories of the robot formation. In both experiments, the formation enters the map from the bottom right corner with <inline-formula id="inf194">
<mml:math id="m210">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>90,90</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> - shown as a red marker, and the red square again shows the formation&#x2019;s final location when the episode ends. The &#x201c;Agent Field End State&#x201d; column shows the final reconstructed field computed by our mobile sensor network. As we can see, the formation center managed to explore the entire map while following an information-rich trajectory, which resulted in consistently high concentration readings and low mapping errors. Additionally, the results from experiments in the high-fidelity simulation environment closely resemble those from the low-fidelity testing environment, demonstrating that our solution can be adapted to more realistic scenarios. Additional screenshots from our laboratory experiments are available in <xref ref-type="fig" rid="F14">Figures 14</xref>, <xref ref-type="fig" rid="F15">15</xref>. The plots in the upper left corners in both figures illustrate the reconstructed fields.</p>
<fig id="F13" position="float">
<label>FIGURE 13</label>
<caption>
<p>The field exploration and reconstruction results in two experiments with two and three diffusion sources. &#x201c;Environment Field End State&#x201d; figures illustrate the end states of the two experiments with corresponding trajectories of the robot formation. The red dots indicate the starting locations of the formation center and the red squares are the ending locations of the formation. &#x201c;Agent Field End State&#x201d; figures show the end states of the reconstructed fields in the two experiments. &#x201c;Concentration&#x201d; figures illustrate the estimated field concentration along the trajectories of the formation center, and &#x201c;Mapping Error&#x201d; figures show the mapping errors while reconstructing the fields.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g013.tif"/>
</fig>
<fig id="F14" position="float">
<label>FIGURE 14</label>
<caption>
<p>Screenshot captured from the experiment on the first environment with two diffusion fields.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g014.tif"/>
</fig>
<fig id="F15" position="float">
<label>FIGURE 15</label>
<caption>
<p>Screenshot captured from the experiment on the second environment with three diffusion fields.</p>
</caption>
<graphic xlink:href="frobt-12-1492526-g015.tif"/>
</fig>
</sec>
<sec id="s7">
<title>7 Conclusions and future work</title>
<p>In this paper, we developed a strategy to map and reconstruct dynamic fields with multiple diffusion sources using a multi-robot formation. This strategy proved effective on various maps with different configurations. Our approach efficiently explores unknown maps while ensuring that potential diffusion sources are detected. The results from our experiments show that the robot formation can effectively utilize environment data from all robots to navigate toward the source center and accurately reconstruct the advection and diffusion coefficients. While we did encounter some challenges with overlapping diffusion fields, these complexities only underscore the need for further research and detailed experiments. Our system holds potential for practical use in scenarios like rescue missions and field explorations, where robots can assess hazards before sending humans into these environments. This research shows the capability and versatility of our multi-robot system in environmental monitoring and could be important in enhancing safety measures during high-risk missions.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s8">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s9">
<title>Author contributions</title>
<p>TL: Writing&#x2013;original draft, Formal Analysis, Methodology, Software, Validation, Visualization, Data curation. DS: Writing&#x2013;original draft, Formal Analysis, Software, Validation, Visualization. DT: Writing&#x2013;review and editing, Methodology, Software. WW: Writing&#x2013;original draft, Conceptualization, Data curation, Formal Analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization.</p>
</sec>
<sec sec-type="funding-information" id="s10">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The research work is supported by NSF grants CMMI-1917300 and RINGS-2148353.</p>
</sec>
<sec sec-type="COI-statement" id="s11">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Agarap</surname>
<given-names>A. F.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Deep learning using rectified linear units (ReLU)</source>. <publisher-name>CoRR abs/1803 08375</publisher-name>. <pub-id pub-id-type="doi">10.48550/arXiv.1803.08375</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bi</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Cure: a hierarchical framework for multi-robot autonomous exploration inspired by centroids of unknown regions</article-title>. <source>IEEE Trans. Automation Sci. Eng.</source> <volume>21</volume> (<issue>3</issue>), <fpage>3773</fpage>&#x2013;<lpage>3786</lpage>. <pub-id pub-id-type="doi">10.1109/tase.2023.3285300</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Moors</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stachniss</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Coordinated multi-robot exploration</article-title>. <source>Robotics, IEEE Trans.</source>
<volume>21</volume>, <fpage>376</fpage>&#x2013;<lpage>386</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2004.839232</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Hma-sar: multi-agent search and rescue for unknown located dynamic targets in completely unknown environments</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>9</volume> (<issue>6</issue>), <fpage>5567</fpage>&#x2013;<lpage>5574</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2024.3396097</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>S.-L.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Adaptive optimal tracking control of an underactuated surface vessel using actor&#x2013;critic reinforcement learning</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>35</volume> (<issue>6</issue>), <fpage>7520</fpage>&#x2013;<lpage>7533</lpage>. <pub-id pub-id-type="doi">10.1109/tnnls.2022.3214681</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Christopoulos</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Roumeliotis</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2005</year>). <source>Adaptive sensing for instantaneous gas release parameter estimation</source>, <fpage>4450</fpage>&#x2013;<lpage>4456</lpage>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Demetriou</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Gatsonis</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Court</surname>
<given-names>J. R.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Coupled controls-computational fluids approach for the estimation of the concentration from a moving gaseous source in a 2-D domain with a Lyapunov-guided sensing aerial vehicle</article-title>. <source>IEEE Trans. Control Syst. Technol.</source> <volume>22</volume> (<issue>3</issue>), <fpage>853</fpage>&#x2013;<lpage>867</lpage>. <pub-id pub-id-type="doi">10.1109/tcst.2013.2267623</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="web">
<collab>Developer Nvidia</collab> (<year>2024</year>). <article-title>Developer Nvidia</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://developer.nvidia.com/embedded/jetson-nano">https://developer.nvidia.com/embedded/jetson-nano</ext-link> (Accessed January 9, 2024)</comment>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dunbabin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Marques</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Robots for environmental monitoring: significant advancements and applications</article-title>. <source>IEEE Robot. Autom. Mag.</source> <volume>19</volume> (<issue>1</issue>), <fpage>24</fpage>&#x2013;<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1109/mra.2011.2181683</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gautam</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shekhawat</surname>
<given-names>V. S.</given-names>
</name>
<name>
<surname>Mohan</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>A graph partitioning approach for fast exploration with multi-robot coordination</article-title>,&#x201d; in <source>2019 IEEE international conference on systems, man and cybernetics (SMC)</source>, <fpage>459</fpage>&#x2013;<lpage>465</lpage>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Graph soft actor&#x2013;critic reinforcement learning for large-scale distributed multirobot coordination</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source>, <fpage>1</fpage>&#x2013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2023.3329530</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Industries</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Adafruit 9-dof absolute orientation imu fusion breakout - bno055</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.adafruit.com/product/4646">https://www.adafruit.com/product/4646</ext-link> (Accessed January 9, 2024)</comment>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khaled</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mustapha</surname>
<given-names>E.-R.</given-names>
</name>
<name>
<surname>Olivier</surname>
</name>
</person-group> (<year>2004</year>). <article-title>On the rate of spread for some reaction-diffusion models of forest fire propagation</article-title>. <source>Numer. Heat. Transf. Part A Appl.</source> <volume>46</volume> (<issue>8</issue>), <fpage>765</fpage>&#x2013;<lpage>784</lpage>. <pub-id pub-id-type="doi">10.1080/104077890504456</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kinaneva</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Hristov</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Raychev</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zahariev</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Early forest fire detection using drones and artificial intelligence</article-title>,&#x201d; in <source>2019 42nd international convention on information and communication technology, electronics and microelectronics</source> <publisher-name>MIPRO Opatija, Croatia</publisher-name>, <fpage>1060</fpage>&#x2013;<lpage>1065</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Adam: a method for stochastic optimization</article-title>,&#x201d; in <source>International conference on learning representations (ICLR)</source> (<publisher-loc>San Diega, CA, USA</publisher-loc>).</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Long</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Multi-robots path planning and mapping for exploring unknown environment</article-title>,&#x201d; in <source>2024 7th international conference on intelligent robotics and control engineering (IRCE)</source>, <fpage>72</fpage>&#x2013;<lpage>76</lpage>.</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Luvisutto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shehhi</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Mankovskii</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Renda</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Stefanini</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>De Masi</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Robotic swarm for marine and submarine missions: challenges and perspectives</article-title>,&#x201d; in <source>2022 IEEE/OES autonomous underwater vehicles symposium (AUV)</source>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Martins</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Almeida</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Almeida</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dias</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dias</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Aaltonen</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Ux 1 system design - a robotic system for underwater mining exploration</article-title>,&#x201d; in <source>IEEE/RSJ international conference on intelligent robots and systems IROS</source>, <fpage>1494</fpage>&#x2013;<lpage>1500</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mourikis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Roumeliotis</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Performance analysis of multirobot cooperative localization</article-title>. <source>IEEE Trans. Robotics</source> <volume>22</volume> (<issue>4</issue>), <fpage>666</fpage>&#x2013;<lpage>681</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2006.878957</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Na</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xumin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yong</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Research on k-means clustering algorithm: an improved k-means clustering algorithm</article-title>,&#x201d; in <source>2010 third international symposium on intelligent information technology and security informatics</source>, <fpage>63</fpage>&#x2013;<lpage>67</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Niroui</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kashino</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Nejat</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Deep reinforcement learning robot for search and rescue applications: exploration in unknown cluttered environments</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>4</volume> (<issue>2</issue>), <fpage>610</fpage>&#x2013;<lpage>617</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2019.2891991</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="web">
<collab>PJRC</collab> (<year>2024</year>). <article-title>Teensy&#xae; 4.0 development board</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://www.pjrc.com/store/teensy40.html">https://www.pjrc.com/store/teensy40.html</ext-link> (Accessed on January 9, 2024)</comment>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Queralta</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Taipalmaa</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Can Pullinen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Sarker</surname>
<given-names>V. K.</given-names>
</name>
<name>
<surname>Nguyen Gia</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tenhunen</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Collaborative multi-robot search and rescue: planning, coordination, perception, and active vision</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>191617</fpage>&#x2013;<lpage>191643</lpage>. <pub-id pub-id-type="doi">10.1109/access.2020.3030190</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Quigley</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Conley</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Gerkey</surname>
<given-names>B. P.</given-names>
</name>
<name>
<surname>Faust</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Foote</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Leibs</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). &#x201c;<article-title>ROS: an open-source robot operating system</article-title>,&#x201d; in <source>ICRA workshop on open source software</source>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Raffin</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Hill</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gleave</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kanervisto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ernestus</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dormann</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Stable-baselines3: reliable reinforcement learning implementations</article-title>. <source>J. Mach. Learn. Res.</source> <volume>22</volume> (<issue>268</issue>), <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.5555/3546258.3546526</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reisch</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Navas-Montilla</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>&#xd6;zgen Xian</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Analytical and numerical insights into wildfire dynamics: exploring the advection&#x2013;diffusion&#x2013;reaction model</article-title>. <source>Comput. Math. Appl.</source> <volume>158</volume>, <fpage>179</fpage>&#x2013;<lpage>198</lpage>. <pub-id pub-id-type="doi">10.1016/j.camwa.2024.01.024</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Beard</surname>
<given-names>R. W.</given-names>
</name>
</person-group> (<year>2008</year>). <source>Distributed consensus in multi-vehicle cooperative control</source>, <volume>27</volume>. <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ribeiro</surname>
<given-names>M. I.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Kalman and extended kalman filters: concept, derivation and properties</article-title>. <source>Inst. Syst. Robotics</source> <volume>43</volume> (<issue>46</issue>), <fpage>3736</fpage>&#x2013;<lpage>3741</lpage>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rossi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Brunelli</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Autonomous gas detection and mapping with unmanned aerial vehicles</article-title>. <source>IEEE Trans. Instrum. Meas.</source> <volume>65</volume> (<issue>4</issue>), <fpage>765</fpage>&#x2013;<lpage>775</lpage>. <pub-id pub-id-type="doi">10.1109/tim.2015.2506319</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Moritz</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Jordan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2015</year>). &#x201c;<article-title>Trust region policy optimization</article-title>,&#x201d; in <source>Proceedings of the 32nd international Conference on international Conference on machine learning - volume 37, Lille, France ICML&#x2019;15</source> (<publisher-name>JMLR.org</publisher-name>), <fpage>1889</fpage>&#x2013;<lpage>1897</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wolski</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klimov</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Proximal policy optimization algorithms</article-title>. <source>CoRR</source> <volume>abs/1707</volume>, <fpage>06347</fpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1707.06347</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shuvo</surname>
<given-names>M. I. R.</given-names>
</name>
<name>
<surname>Wimer</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Mahmud</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.-H.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>A novel collaborative knowledge sharing and self-learning framework for robotic systems in search and rescue operations</article-title>,&#x201d; in <source>IECON 2023- 49th annual conference of the IEEE industrial electronics society</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Talwar</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Deep reinforcement learning based path-planning for multi-agent systems in advection-diffusion field reconstruction tasks</source>.</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tricaud</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y. Q.</given-names>
</name>
</person-group> (<year>2010</year>). &#x201c;<article-title>Optimal trajectories of mobile remote sensors for parameter estimation in distributed cyber-physical systems</article-title>,&#x201d; in <source>Proceedings of the 2010 American control conference</source>, <fpage>3211</fpage>&#x2013;<lpage>3216</lpage>.</citation>
</ref>
<ref id="B35">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ucinski</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Optimal measurement methods for distributed parameter system identification</article-title>,&#x201d; in <source>Taylor and Francis series in systems and control</source>. <publisher-loc>Boca Raton, Fla</publisher-loc>: <publisher-name>CRC Press</publisher-name>.</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Ucinski</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2005</year>). &#x201c;<article-title>Time-optimal path planning of moving sensors for parameter estimation of distributed systems</article-title>,&#x201d; in <source>Proceedings of the 44th IEEE conference on decision and control</source>, <fpage>5257</fpage>&#x2013;<lpage>5262</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>H.-C.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S.-C.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>P.-J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.-L.</given-names>
</name>
<name>
<surname>Teng</surname>
<given-names>Y.-C.</given-names>
</name>
<name>
<surname>Ko</surname>
<given-names>Y.-T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Curriculum reinforcement learning from avoiding collisions to navigating among movable obstacles in diverse environments</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>8</volume> (<issue>5</issue>), <fpage>2740</fpage>&#x2013;<lpage>2747</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2023.3251193</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A survey on curriculum learning</article-title>. <source>IEEE Trans. Pattern Analysis Mach. Intell.</source> <volume>44</volume> (<issue>9</issue>), <fpage>4555</fpage>&#x2013;<lpage>4576</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2021.3069908</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>You</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Parameter identification of spatial&#x2013;temporal varying processes by a multi-robot system in realistic diffusion fields</article-title>. <source>Robotica</source> <volume>39</volume>, <fpage>842</fpage>&#x2013;<lpage>861</lpage>. <pub-id pub-id-type="doi">10.1017/s0263574720000788</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Robust cooperative exploration with a switching strategy</article-title>. <source>IEEE Trans. Robotics</source> <volume>28</volume> (<issue>4</issue>), <fpage>828</fpage>&#x2013;<lpage>839</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2012.2190182</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="web">
<collab>YDLIDAR-G4-Datasheet</collab> (<year>2024</year>). <article-title>YDLIDAR-G4-Datasheet</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="http://www.ydlidar.com/Public/upload/files/2020-04-13/YDLIDAR%20G4%20Datasheet.pdf">http://www.ydlidar.com/Public/upload/files/2020-04-13/YDLIDAR%20G4%20Datasheet.pdf</ext-link> (Accessed on January 9, 2024)</comment>.</citation>
</ref>
<ref id="B42">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>You</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Geometric reinforcement learning based path planning for mobile sensor networks in advection-diffusion field reconstruction</article-title>,&#x201d; in <source>2018 IEEE conference on decision and control (CDC)</source> (<publisher-name>IEEE</publisher-name>), <fpage>1949</fpage>&#x2013;<lpage>1954</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>You</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Cooperative filtering for parameter identification of diffusion processes</article-title>,&#x201d; in <source>2016 IEEE 55th conference on decision and control</source> <publisher-loc>IEEE</publisher-loc>: <publisher-name>CDC</publisher-name>, <fpage>4327</fpage>&#x2013;<lpage>4333</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>You</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Cooperative filtering and parameter identification for advection&#x2013;diffusion processes using a mobile sensor network</article-title>. <source>IEEE Trans. Control Syst. Technol.</source> <volume>31</volume> (<issue>2</issue>), <fpage>527</fpage>&#x2013;<lpage>542</lpage>. <pub-id pub-id-type="doi">10.1109/tcst.2022.3183585</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Leonard</surname>
<given-names>N. E.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Cooperative filters and control for cooperative exploration</article-title>. <source>IEEE Trans. Automatic Control</source> <volume>55</volume> (<issue>3</issue>), <fpage>650</fpage>&#x2013;<lpage>663</lpage>. <pub-id pub-id-type="doi">10.1109/tac.2009.2039240</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mayberry</surname>
<given-names>S. T.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Distributed cooperative kalman filter constrained by advection&#x2013;diffusion equation for mobile sensor networks</article-title>. <source>Front. Robotics AI</source> <volume>10</volume>, <fpage>1175418</fpage>. <pub-id pub-id-type="doi">10.3389/frobt.2023.1175418</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Multi-robot autonomous exploration in unknown environment: a review</article-title>,&#x201d; in <source>2023 China automation congress (CAC)</source>, <fpage>7639</fpage>&#x2013;<lpage>7644</lpage>.</citation>
</ref>
</ref-list>
</back>
</article>