<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article article-type="methods-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Future Transp.</journal-id>
<journal-title>Frontiers in Future Transportation</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Future Transp.</abbrev-journal-title>
<issn pub-type="epub">2673-5210</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1524232</article-id>
<article-id pub-id-type="doi">10.3389/ffutr.2025.1524232</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Future Transportation</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Reinforcement learning based estimation of shortest paths in dynamically changing transportation networks</article-title>
<alt-title alt-title-type="left-running-head">Pham et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/ffutr.2025.1524232">10.3389/ffutr.2025.1524232</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Pham</surname>
<given-names>Hoang Dat</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2889275/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Narasimhamurthy</surname>
<given-names>Sharath Mysore</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1303724/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Mehran</surname>
<given-names>Babak</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1382333/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Manley</surname>
<given-names>Ed</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ashraf</surname>
<given-names>Ahmed</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1913251/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Electrical and Computer Engineering</institution>, <institution>University of Manitoba</institution>, <addr-line>Winnipeg</addr-line>, <addr-line>MB</addr-line>, <country>Canada</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Civil Engineering</institution>, <institution>University of Manitoba</institution>, <addr-line>Winnipeg</addr-line>, <addr-line>MB</addr-line>, <country>Canada</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Geography</institution>, <institution>University of Leeds</institution>, <addr-line>Leeds</addr-line>, <country>United Kingdom</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2191516/overview">Dongyao Jia</ext-link>, Xi&#x2019;an Jiaotong-Liverpool University, China</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/973323/overview">David L&#xf3;pez Flores</ext-link>, National Autonomous University of Mexico, Mexico</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2912599/overview">Weizhen Han</ext-link>, Wuhan University of Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Ahmed Ashraf, <email>ahmed.ashraf@umanitoba.ca</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>06</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>6</volume>
<elocation-id>1524232</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Pham, Narasimhamurthy, Mehran, Manley and Ashraf.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Pham, Narasimhamurthy, Mehran, Manley and Ashraf</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Finding the shortest path in a network is a classical problem, and a variety of search strategies have been proposed to solve it. In this paper, we review traditional approaches for finding shortest paths, namely, uninformed search, informed search and incremental search. The above traditional algorithms have been put to successful use for fixed networks with static link costs. However, in many practical contexts, such as transportation networks, the link costs can vary over time. We investigate the applicability of the aforementioned benchmark search strategies in a simulated transportation network where link costs (travel times) are dynamically estimated with vehicle mean speeds. As a comparison, we present performance metrics for a reinforcement learning based routing algorithm, which can interact with the network and learn the changing link costs through experience. Our results suggest that reinforcement learning algorithm computes optimal paths dynamically.</p>
</abstract>
<kwd-group>
<kwd>shortest path</kwd>
<kwd>reinforcement learning</kwd>
<kwd>transportation network</kwd>
<kwd>dijkstra</kwd>
<kwd>A&#x2217;</kwd>
<kwd>dynamic link cost</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Connected Mobility and Automation</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Autonomous transportation is expected to become a prevalent means of public transportation in the near future (<xref ref-type="bibr" rid="B12">Hancock et al., 2019</xref>). Such a transportation system demands that algorithms for estimating shortest paths be well-adapted to the continuously changing nature of the traffic network. In a stochastically evolving traffic network, the link costs (vehicular travel times in this study) can dynamically change and are constantly influenced by a range of traffic factors such as traffic congestion, road work, and bad weather, among others. As a result, an autonomous transit system fundamentally depends on the availability of a real-time routing system capable of estimating the shortest path not only before the trip, but also of adaptive rerouting in response to fluctuating link costs.</p>
<p>Over the past decades, the shortest path problem has been extensively investigated with applications ranging from computer networks to transportation networks and many approaches have been explored to examine the effectiveness of search strategies. In this respect, some of the traditional search strategies include uninformed search, informed search, and incremental search (<xref ref-type="bibr" rid="B22">Madkour et al., 2017</xref>; <xref ref-type="bibr" rid="B16">Katre and Thakare, 2017</xref>; <xref ref-type="bibr" rid="B35">Surekha and Santosh, 2016</xref>; <xref ref-type="bibr" rid="B23">Magzhan and Jani, 2013</xref>; <xref ref-type="bibr" rid="B43">Zhan and Noon, 1998</xref>; <xref ref-type="bibr" rid="B42">Zhan, 1997</xref>; <xref ref-type="bibr" rid="B11">Fua and Rilett, 2006</xref>; <xref ref-type="bibr" rid="B34">Sunita Kumawat and Kumar, 2021</xref>; <xref ref-type="bibr" rid="B27">Pallottino and Scutella, 1998</xref>; <xref ref-type="bibr" rid="B15">Huang et al., 2007</xref>; <xref ref-type="bibr" rid="B22">Madkour et al., 2017</xref>; <xref ref-type="bibr" rid="B16">Katre and Thakare, 2017</xref>; <xref ref-type="bibr" rid="B35">Surekha and Santosh, 2016</xref>; <xref ref-type="bibr" rid="B23">Magzhan and Jani, 2013</xref>; <xref ref-type="bibr" rid="B43">Zhan and Noon, 1998</xref>; <xref ref-type="bibr" rid="B42">Zhan, 1997</xref>; <xref ref-type="bibr" rid="B11">Fua and Rilett, 2006</xref>; <xref ref-type="bibr" rid="B34">Sunita Kumawat and Kumar, 2021</xref>; <xref ref-type="bibr" rid="B27">Pallottino and Scutella, 1998</xref>; <xref ref-type="bibr" rid="B15">Huang et al., 2007</xref>). The above strategies have been shown to be very successful for problem scenarios involving fixed networks with static link costs over time. However, they present several limitations in terms of four criteria that we discuss later in this paper, especially when the traffic network is dynamically changing or has missing information. Another class of search strategies, namely, reinforcement learning (RL) based search, can adapt to changing link costs. RL based methods have been primarily applied in computer networks, and their potential for transportation networks has started to get attention only recently. Applications of reinforcement learning have been reviewed comprehensively in review papers (<xref ref-type="bibr" rid="B9">Farazi et al., 2021</xref>), and the most relevant studies are for vehicle routing optimization, known in other terms as traveling salesman problems, yet these studies lack the research on reinforcement learning for finding the shortest path. Therefore, a comparison between traditional and RL-based methods for computing optimal shortest paths should be conducted.</p>
<p>In particular, the method that we develop is based on Q-learning (<xref ref-type="bibr" rid="B36">Sutton and Barto, 2018</xref>). For the task of path optimization, Q-learning has been used for communication networks (<xref ref-type="bibr" rid="B4">Boyan and Littman, 1994</xref>). There has been use of Q-learning for mobile communication in Vehicular Ad hoc networks (VANETS) wherein a moving vehicle is considered as a mobile node in a wireless network (<xref ref-type="bibr" rid="B20">Li et al., 2014</xref>)<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref>. For transportation networks, there is significant body of work on Q-learning based intelligent traffic signal control (<xref ref-type="bibr" rid="B5">Chin et al., 2012</xref>; <xref ref-type="bibr" rid="B8">Ducrocq and Farhi, 2023</xref>; <xref ref-type="bibr" rid="B26">Moreno-Malo et al., 2024</xref>; <xref ref-type="bibr" rid="B5">Chin et al., 2012</xref>; <xref ref-type="bibr" rid="B8">Ducrocq and Farhi, 2023</xref>; <xref ref-type="bibr" rid="B26">Moreno-Malo et al., 2024</xref>). Moreover, Q-learning has been employed for reducing traffic congestion (<xref ref-type="bibr" rid="B37">Swapno et al., 2024</xref>). There is one work related to loop-breaking in route-planning through Q-learning in vehicular networks (<xref ref-type="bibr" rid="B25">Meerhof, 2021</xref>). As such the potential of Q-learning to perform path optimization in dynamically changing vehicular/transportation networks remains underexplored.The main contributions of this paper are summarized as below:<list list-type="simple">
<list-item>
<p>&#x2022; We investigate the performance of traditional shortest path algorithms (such as Dijkstra and <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>) in simulated transportation network where link costs are dynamically estimated and highly correlated based on the current traffic condition.</p>
</list-item>
<list-item>
<p>&#x2022; We investigate the use of the reinforcement search strategy, namely, Q-routing (<xref ref-type="bibr" rid="B4">Boyan and Littman, 1994</xref>) in the same transportation network and compare its performance to that of other search strategies.</p>
</list-item>
</list>
</p>
<p>In this paper, a transportation network is simulated using VISSIM traffic microsimulation environment (<xref ref-type="bibr" rid="B10">Fellendorf and Vortisch, 2020</xref>). Traffic obstructions are introduced into the network to induce fluctuations in vehicle volumes and speeds. These vehicle speeds are converted into link costs and are used as weights for shortest path algorithms.</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>The problem of shortest path estimation, which has been widely researched, is a principal and classical challenge in transportation and computer networks. All shortest path algorithms investigated in this paper will treat a transportation network as a directed graph with travel time as non-negative link cost. Shortest path algorithms can be evaluated based on the following four criteria:<list list-type="simple">
<list-item>
<p>&#x2022; Completeness: determines whether or not the algorithm is guaranteed to find the solution to the problem, if one exists.</p>
</list-item>
<list-item>
<p>&#x2022; Optimality: evaluates whether the solution provided by the algorithm is the best.</p>
</list-item>
<list-item>
<p>&#x2022; Time complexity: evaluates how long the algorithm takes to solve the problem. Usually, it is expressed in the big O notation representing the order of growth in computational time as the number of inputs grows.</p>
</list-item>
<list-item>
<p>&#x2022; Space complexity: evaluates how much memory space the algorithm consumed to reach the final solution.</p>
</list-item>
</list>
</p>
<sec id="s2-1">
<title>2.1 Search strategies</title>
<p>Traditional AI literature (<xref ref-type="bibr" rid="B33">Russell and Norvig, 2003</xref>) distinguishes three types of search strategies:<list list-type="simple">
<list-item>
<p>&#x2022; Uninformed search i.e., the algorithm employs no method for estimating how close the search process is to a destination.</p>
</list-item>
<list-item>
<p>&#x2022; Informed search i.e., the algorithm employs heuristics to direct the search to its destination.</p>
</list-item>
<list-item>
<p>&#x2022; Incremental search i.e., the algorithm reuses information from previous searches to find updated shortest path solutions faster, thus, eliminating the need to search for the shortest path from scratch.</p>
</list-item>
</list>
</p>
<p>In addition to the above, RL has also been used for estimating shortest paths by exploring a transportation network wherein the travel times experienced while executing a path are considered as negative rewards.</p>
<sec id="s2-1-1">
<title>2.1.1 Uninformed search</title>
<p>Uninformed search strategies refer to a group of search techniques wherein the algorithm has no access to any information regarding how far the goal state is, though the algorithm can check if the current state is a goal state or not. It is also known as blind search. <xref ref-type="table" rid="T1">Table 1</xref> summarizes the current uninformed search techniques (<xref ref-type="bibr" rid="B33">Russell and Norvig, 2003</xref>), including breadth-first search, depth-first search, depth-limited search, iterative deepening search, uniform cost search, bidirectional search, Dijkstra (<xref ref-type="bibr" rid="B7">Dijkstra, 1959</xref>), and their performance criteria. Among uninformed search methods, the Dijkstra algorithm is considered as benchmark both in terms of time and space complexity since its complexity does not depend either on the branching factor or the depth of the solution, but only depends on the number of nodes in the network.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Uninformed search, informed search (<xref ref-type="bibr" rid="B33">Russell and Norvig, 2003</xref>) and incremental search.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Completeness</th>
<th align="left">Optimality</th>
<th align="left">Time complexity</th>
<th align="left">Space complexity</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td colspan="5" align="left">Uninformed Search</td>
</tr>
<tr>
<td align="left">Breadth-first</td>
<td align="left">Yes</td>
<td align="left">Yes</td>
<td align="left">
<inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Depth-first</td>
<td align="left">Yes</td>
<td align="left">No</td>
<td align="left">
<inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Depth-limited</td>
<td align="left">No</td>
<td align="left">No</td>
<td align="left">
<inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Iterative Deepening</td>
<td align="left">Yes</td>
<td align="left">Yes</td>
<td align="left">
<inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Uniform cost</td>
<td align="left">Yes</td>
<td align="left">Yes</td>
<td align="left">
<inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Bidirectional</td>
<td align="left">Yes</td>
<td align="left">Yes</td>
<td align="left">
<inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Dijkstra</td>
<td align="left">Yes</td>
<td align="left">Yes</td>
<td align="left">
<inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf15">
<mml:math id="m15">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td colspan="5" align="left">Informed Search</td>
</tr>
<tr>
<td align="left">Best-first</td>
<td align="left">No</td>
<td align="left">No</td>
<td align="left">
<inline-formula id="inf16">
<mml:math id="m16">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf17">
<mml:math id="m17">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Greedy best-first</td>
<td align="left">No</td>
<td align="left">No</td>
<td align="left">
<inline-formula id="inf18">
<mml:math id="m18">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf19">
<mml:math id="m19">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">A&#x2a;</td>
<td align="left">Yes<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="left">Yes<xref ref-type="table-fn" rid="Tfn1">
<sup>a</sup>
</xref>
</td>
<td align="left">
<inline-formula id="inf20">
<mml:math id="m20">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf21">
<mml:math id="m21">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">Hill Climbing</td>
<td align="left">No</td>
<td align="left">No</td>
<td align="left">
<inline-formula id="inf22">
<mml:math id="m22">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x221e;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf23">
<mml:math id="m23">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td colspan="5" align="left">Incremental Search</td>
</tr>
<tr>
<td align="left">DynamicSWSF-FP <xref ref-type="bibr" rid="B30">Ramalingam and Reps, (1996a)</xref>
</td>
<td align="left">Yes</td>
<td align="left">Yes</td>
<td align="left">
<inline-formula id="inf24">
<mml:math id="m24">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mo>.</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf25">
<mml:math id="m25">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mfenced open="(" close="">
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>b is branching factor, d is depth of the solution, m is maximum depth of tree, l is depth limit of tree, C is cost of optimal solution, <inline-formula id="inf26">
<mml:math id="m26">
<mml:mrow>
<mml:mo stretchy="false">&#x2016;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mo stretchy="false">&#x2016;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> is a measure related to &#x201c;the size of the change in the input and output&#x201d;, <inline-formula id="inf27">
<mml:math id="m27">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>&#x3b4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is a bound on the time required to compute the function associated with any vertex in <italic>Changed U Succ(Changed)</italic> (<xref ref-type="bibr" rid="B30">Ramalingam and Reps, 1996a</xref>), <italic>e</italic> is each step closer to goal node and <italic>n</italic> is number of nodes.</p>
</fn>
<fn id="Tfn1">
<label>
<sup>a</sup>
</label>
<p>if heuristic function is admissible and monotonic.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s2-1-2">
<title>2.1.2 Informed (heuristic) search</title>
<p>The primary benefit of heuristic-based strategies is that the search space can be narrowed down. Many shortest path algorithms with a reduced search space have been proposed, such as best-first search (<xref ref-type="bibr" rid="B28">Pearl, 1984</xref>), greedy best-first search, hill climbing search, and A&#x2a; (<xref ref-type="bibr" rid="B13">Hart et al., 1968</xref>) as shown in <xref ref-type="table" rid="T1">Table 1</xref>. These types of search methods attempt to narrow down the search space by utilizing various sources of additional information. Also, the A&#x2a; algorithm is complete and optimal only if its heuristic function is admissible and monotonic, and many modern algorithms such as D&#x2a; lite (<xref ref-type="bibr" rid="B18">Koenig and Likhachev, 2002</xref>) and LPA&#x2a; (<xref ref-type="bibr" rid="B17">Koenig and Likhachev, 2001</xref>) are based on A&#x2a;. Therefore, the A&#x2a; algorithm stands out in the class of informed search because of its completeness and optimality.</p>
</sec>
<sec id="s2-1-3">
<title>2.1.3 Incremental search</title>
<p>Incremental search strategies are applicable for networks in which only a small number of link costs are likely to change at a time. Since these changes affect only a part of the graph, recomputing shortest paths for the entire graph is not necessary. Instead, it is possible to update paths corresponding to only a subset of the graph. As a result, such methods are used to solve dynamic shortest path problems, which require determining shortest paths repeatedly as the topology of a graph or its link costs only change partially. A representative incremental search algorithm is the Ramalingam and Reps&#x2019; algorithm (<xref ref-type="bibr" rid="B31">Ramalingam and Reps, 1996b</xref>) (RR), also known as the DynamicSWSF-FP algorithm. However, if all the link costs in the graph change, incremental search algorithms are unable to capitalize on previous search results.</p>
<sec id="s2-1-3-1">
<title>2.1.3.1 DynamicSWSF-FP Algorithm</title>
<p>In dynamic transportation networks, often, only a portion of links change in terms of cost between updates. Starting costs for some of the nodes remain unchanged and thus do not need to be recalculated. As such, recomputing all the optimal routes could be wasteful since some of the previous search results can be reused. Incremental search methods, such as the RR algorithm, reuse information from previous searches to find the shortest paths for a series of similar path-planning problems, which is faster than solving each path-planning problem from the scratch. A key aspect about reusing previous search results is determining which costs have been affected by the cost update operation and must be recalculated. The RR algorithm employs two estimates: one that corresponds directly to starting distance in Dijkstra&#x2019;s algorithm and another one is a right-hand-side value (<italic>rhs</italic>) for checking the local consistency to prevent path recalculation (<xref ref-type="bibr" rid="B31">Ramalingam and Reps, 1996b</xref>).</p>
</sec>
</sec>
<sec id="s2-1-4">
<title>2.1.4 Reinforcement learning search</title>
<p>The field of RL has grown significantly over the past decades, with applications ranging from robotics (<xref ref-type="bibr" rid="B29">Polydoros and Nalpantidis, 2017</xref>), communications (<xref ref-type="bibr" rid="B21">Luong et al., 2019</xref>) to gaming AI (<xref ref-type="bibr" rid="B40">Vinyals et al., 2019</xref>). Among RL algorithms, Q-routing is a method that was proposed primarily for communication networks but can also be used in transportation systems (<xref ref-type="bibr" rid="B4">Boyan and Littman, 1994</xref>). Q-routing is based on Q-learning (<xref ref-type="bibr" rid="B36">Sutton and Barto, 2018</xref>) and does not need to know the link costs to start off, and can learn optimal paths over time through experience. In particular, for every node <inline-formula id="inf30">
<mml:math id="m30">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in the graph, a 2D Q-table, <inline-formula id="inf31">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, is maintained that contains the cost of transitioning from <inline-formula id="inf32">
<mml:math id="m32">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to node <inline-formula id="inf33">
<mml:math id="m33">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> (where <inline-formula id="inf34">
<mml:math id="m34">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is one of the neighbors of node <inline-formula id="inf35">
<mml:math id="m35">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., <inline-formula id="inf36">
<mml:math id="m36">
<mml:mrow>
<mml:mfenced open="" close=")">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>, if the final destination is node <inline-formula id="inf37">
<mml:math id="m37">
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. During a training episode, at each step, the next move is chosen based on the current values in the Q-table, the move is executed, and the actual time experienced while executing the move is stored. Based on this experience, a Bellman update of the Q-values is performed as follows:<disp-formula id="e1">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b7;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
<mml:mtext>&#x2003;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf38">
<mml:math id="m39">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the next move chosen based on current Q-values, i.e.,<disp-formula id="e2">
<mml:math id="m40">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi mathvariant="italic">argmin</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:munder>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
</p>
<p>In <xref ref-type="disp-formula" rid="e1">Equation 1</xref>, <inline-formula id="inf39">
<mml:math id="m41">
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> belong to the list of the adjacent nodes of <inline-formula id="inf40">
<mml:math id="m42">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf41">
<mml:math id="m43">
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the step-size, and the <inline-formula id="inf42">
<mml:math id="m44">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the time experienced estimated to go from <inline-formula id="inf43">
<mml:math id="m45">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf44">
<mml:math id="m46">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> which serves as a surrogate for the link cost (reward). In other words, if Q-values are consistent, then <inline-formula id="inf45">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>y&#x2a;</italic> is computed using <xref ref-type="disp-formula" rid="e1">Equation 1</xref>, should be equal to the time estimated from <inline-formula id="inf46">
<mml:math id="m48">
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf47">
<mml:math id="m49">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> plus the cost associated with the best move thereafter, and the update is proportional to the difference from this desired value.</p>
<p>As shown in <xref ref-type="table" rid="T2">Table 2</xref>, Q-routing requires less run-time complexity as it is executed more often. During training mode or offline mode, it takes <inline-formula id="inf48">
<mml:math id="m50">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> complexity to run the Q-routing algorithm, where T is the number of training loops, n is the number of nodes and b is branching factor. But, time complexity is reduced to <inline-formula id="inf49">
<mml:math id="m51">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> during online mode.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Reinforcement Learning search.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">RL search</th>
<th align="left">Completeness</th>
<th align="left">Optimality</th>
<th align="left">Time complexity</th>
<th align="left">Space complexity</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Q-routing</td>
<td align="left">Yes</td>
<td align="left">Yes</td>
<td align="left">
<inline-formula id="inf28">
<mml:math id="m28">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>n</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">
<inline-formula id="inf29">
<mml:math id="m29">
<mml:mrow>
<mml:mi>O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>T is the number of training loops, b is the branching factor and n is number of nodes.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s3">
<title>3 Traffic network simulation</title>
<sec id="s3-1">
<title>3.1 Traffic network description</title>
<p>A simple transportation network in the form of square grids is used in this study as shown in <xref ref-type="fig" rid="F1">Figure 1A</xref>. The network has 25 nodes and 80 links (each 2-lane and directed).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Details of the toy network used in the study. <bold>(A)</bold> Network shown in the form 5 x 5 square grids. <bold>(B, C)</bold>: Signal heads and feasible movements at each intersection.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g001.tif"/>
</fig>
<p>We have chosen to simulate a modestly sized network because of the following reasons. Dynamic rerouting of vehicles is highly influenced by accurate link travel cost estimations, which depend on interactions between vehicles. As such, microscopic traffic simulation is critical for capturing localized congestion effects. Smaller &#x201c;toy networks&#x201d; are commonly used for developing and validating methodologies [e.g. (<xref ref-type="bibr" rid="B41">Wang et al., 2022</xref>; <xref ref-type="bibr" rid="B32">Rampf et al., 2023</xref>; <xref ref-type="bibr" rid="B3">Bhavsar et al., 2014</xref>; <xref ref-type="bibr" rid="B19">Koh et al., 2020</xref>; <xref ref-type="bibr" rid="B41">Wang et al., 2022</xref>; <xref ref-type="bibr" rid="B32">Rampf et al., 2023</xref>; <xref ref-type="bibr" rid="B3">Bhavsar et al., 2014</xref>; <xref ref-type="bibr" rid="B19">Koh et al., 2020</xref>)]. Notably, variants of the Q-routing algorithm, such as the one developed in this study, are extensively utilized in large-scale communication networks [e.g. (<xref ref-type="bibr" rid="B24">Mammeri, 2019</xref>; <xref ref-type="bibr" rid="B1">Alam and Moh, 2022</xref>; <xref ref-type="bibr" rid="B2">Al-Rawi et al., 2015</xref>; <xref ref-type="bibr" rid="B24">Mammeri, 2019</xref>; <xref ref-type="bibr" rid="B1">Alam and Moh, 2022</xref>; <xref ref-type="bibr" rid="B2">Al-Rawi et al., 2015</xref>)] and have demonstrated robust performance. Given that transportation networks are typically much smaller and less complex than communication networks, at this time we will be validating the methods on smaller networks and explore larger scales in later studies.</p>
<p>In the 25-node network simulated in the current study, the length of each link is 2,500&#xa0;m. To better capture heterogeneous flow conditions along these relatively long links, we split them into segments. Based on sensitivity analysis, we determined that a segment length of 25&#xa0;m provides a reliable representation of link travel times. Further reducing the segment length does not significantly improve accuracy, and hence, links are divided into 100 equal-length segments. While our study uses equal-length segments for simplicity and consistency, we employed a general formula for computing weighted average speeds to account for potential scenarios where varying segment lengths might be necessary. For example, unequal segment lengths could be useful to represent varying geometries, such as curves or bottlenecks, along a link. However, in this study, such complexities were not required, and our approach ensures both computational efficiency and practical applicability.</p>
<p>The network is composed of nine signalized intersections. The signal heads and feasible movements at each intersection are shown in <xref ref-type="fig" rid="F1">Figures 1B, C</xref>. Right turning movements are permitted on red. Through and left turning movements are simultaneously allowed (but controlled by the traffic signal). The node annotations and coordinates are shown in <xref ref-type="fig" rid="F2">Figure 2</xref>. The four signal heads at an intersection are sequentially operated with a cycle length of 240&#xa0;s (57s green &#x2b; 3s amber per signal head). All the nine traffic signals are synchronized.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Node numbers and coordinates.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g002.tif"/>
</fig>
<p>The demand for travel arises at two main origin intersections, <inline-formula id="inf50">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>-<inline-formula id="inf51">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf52">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>-<inline-formula id="inf53">
<mml:math id="m55">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which are located at nodes one and 5. Two main destination intersections, <inline-formula id="inf54">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>-<inline-formula id="inf55">
<mml:math id="m57">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf56">
<mml:math id="m58">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>-<inline-formula id="inf57">
<mml:math id="m59">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, which are located at nodes 25 and 21 are considered. Coordinates for every intersection of origin and destination can be viewed in <xref ref-type="fig" rid="F2">Figure 2</xref>. The magnitude of travel demand between OD pairs <inline-formula id="inf58">
<mml:math id="m60">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mspace width="0.3333em"/>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mo>&#x2200;</mml:mo>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
<mml:mtext>&#x2003;</mml:mtext>
<mml:mtext>such&#x2009;that&#x2009;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is taken as 500 vehicles per hour. Such a large value of travel demand is considered to ensure a build-up of queue/congestion in the network. Besides main OD pairs that have large travel demand, other random OD pairs are also used to run and evaluate different shortest path algorithms.</p>
<p>The aforementioned network is modeled in PTV VISSIM, which is a microscopic traffic simulator. As the size of a network increases, it becomes very tedious to determine and assign routes (sequence of links) for vehicles between several origin-destination pairs. Also, a predetermined route assignment in a simulation study does not reflect the route choices made by drivers in the real world (<xref ref-type="bibr" rid="B10">Fellendorf and Vortisch, 2020</xref>). Hence, the route choices are dynamically made in this study. The ability of VISSIM to compute dynamic stochastic user equilibrium is exploited to determine the routes dynamically. A comprehensive representation of route choice is thus possible. The vehicle composition of 90% cars and 10% heavy vehicles is used. Other parameters for the network simulation are as follows. The desired speed distribution used is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>. For the driving behavior model, the Wiedemann-74 car-following model is employed as it is considered suitable in urban environment. The default Car-following parameters used are:<list list-type="simple">
<list-item>
<p>&#x2022; Average standstill distance: 2&#xa0;m</p>
</list-item>
<list-item>
<p>&#x2022; Additive part of safety distance: 2</p>
</list-item>
<list-item>
<p>&#x2022; Multiplicative part of safety distance: three</p>
</list-item>
</list>
</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Speed distribution.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g003.tif"/>
</fig>
<p>VISSIM also has a feature termed &#x201c;reduced speed areas&#x201d;, which is used to create temporary or localized congestion (e.g., a traffic incident). Three links are randomly chosen to introduce congestion as shown in <xref ref-type="table" rid="T3">Table 3</xref>. A simulation step size of 0.1s and a simulation period of 2&#xa0;h are adopted. After every 100 simulation steps, the state of every vehicle in the network is sampled (and stored for further analysis) by using VISSIM COM. The state of a vehicle includes the link on which a vehicle is located, position on that link, lane occupancy, instantaneous speed, acceleration, and vehicle type.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Congested links.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Link</th>
<th align="center">Length of congestion zone (m)</th>
<th align="center">Desired speed (km/h)</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">26</td>
<td align="center">500</td>
<td align="center">20</td>
</tr>
<tr>
<td align="center">10</td>
<td align="center">200</td>
<td align="center">12</td>
</tr>
<tr>
<td align="center">34</td>
<td align="center">300</td>
<td align="center">12</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The desired speed distribution within the congestion zone is as shown in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Speed distribution within the congestion zone.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g004.tif"/>
</fig>
</sec>
<sec id="s3-2">
<title>3.2 Ground-truth link costs from simulation</title>
<p>To reliably measure ground-truth dynamic links costs we need to keep track of the spatial and temporal variation of link costs (travel times). The state of all the vehicles in the network is extracted from VISSIM after every 100 simulation steps. Every link between nodes <italic>i</italic> and <italic>j</italic> is considered to be composed of <inline-formula id="inf59">
<mml:math id="m61">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> segments (need not be of equal length) as shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. The length of the <inline-formula id="inf60">
<mml:math id="m62">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>-th segment of link <inline-formula id="inf61">
<mml:math id="m63">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is <inline-formula id="inf62">
<mml:math id="m64">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and total length of the link is <inline-formula id="inf63">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Segmentation of a link <inline-formula id="inf64">
<mml:math id="m66">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g005.tif"/>
</fig>
<p>The space mean speed of the segment <inline-formula id="inf65">
<mml:math id="m67">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at a time step <inline-formula id="inf66">
<mml:math id="m68">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>M</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> can be computed from the extracted states. The cost of traveling on link <inline-formula id="inf67">
<mml:math id="m69">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> at a time step <inline-formula id="inf68">
<mml:math id="m70">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is then computed as in <xref ref-type="disp-formula" rid="e3">Equation 3</xref>:<disp-formula id="e3">
<mml:math id="m71">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>M</mml:mi>
<mml:msubsup>
<mml:mrow>
<mml:mi>S</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
<mml:mspace width="2em"/>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2200;</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:munder>
</mml:mstyle>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where, <inline-formula id="inf69">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula> is the weight for segment <inline-formula id="inf70">
<mml:math id="m73">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Desired speed is considered as the space mean speed of a segment not hosting any vehicle.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Algorithm performance</title>
<sec id="s4-1">
<title>4.1 Algorithm selection</title>
<p>From the related literature, it is clear that Dijkstra and <inline-formula id="inf71">
<mml:math id="m74">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> are the benchmark candidates and representative algorithms for uninformed and informed search respectively in terms of performance criteria. In communication networks, algorithms such as distance vector routing (<xref ref-type="bibr" rid="B14">Hedrick, 1988</xref>) have been employed. Such algorithms rely on continuous signal transmission between nodes to update travel time, thus they are deemed inapplicable in our transportation network where travel time is estimated at each time step based on the number of physically present vehicles. Moreover, link costs in our simulated networks depend on the number of vehicles and their speeds. As the number of vehicles entering a link changes, and vehicle speeds vary, link costs dynamically change at every query time for all links. As a result, the RR algorithms and similar incremental algorithms such as Lifelong planning A&#x2a; (<xref ref-type="bibr" rid="B17">Koenig and Likhachev, 2001</xref>) and D&#x2a; Lite (<xref ref-type="bibr" rid="B18">Koenig and Likhachev, 2002</xref>) based on it cannot benefit from incremental search because the estimate changes dynamically, and thus, local consistency will never be met. Therefore, we chose some of the basic algorithms that best represent their classes, namely, Dijkstra for uninformed search, A&#x2a; for informed search, and Q-routing for reinforcement learning, and compare their performances in our case study network.</p>
<p>
<xref ref-type="fig" rid="F6">Figures 6</xref>&#x2013;<xref ref-type="fig" rid="F8">8</xref> show the pseudocode for each algorithm. R1.5 The heuristic function that we use in the <inline-formula id="inf72">
<mml:math id="m75">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> algorithm is the Manhattan distance function that estimates how close the current node is to the destination based on the nodes&#x2019; coordinates. Manhattan distance has been used because it is better represents a real-world scenario with grid-networks.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Dijkstra&#x2019;s algorithm.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g006.tif"/>
</fig>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Q-routing algorithm.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g007.tif"/>
</fig>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>A&#x2a; algorithm.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g008.tif"/>
</fig>
</sec>
<sec id="s4-2">
<title>4.2 Algorithm performance</title>
<sec id="s4-2-1">
<title>4.2.1 Static network</title>
<p>We first test the Dijkstra and A&#x2a; algorithms on our simulated transportation network when there is no vehicle. The purpose of this test is to check the performance of the heuristic function in <inline-formula id="inf73">
<mml:math id="m76">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. When there is no traffic in the network, the link costs are the link lengths and the network is static. <xref ref-type="fig" rid="F9">Figure 9</xref> shows the identical performance of <inline-formula id="inf74">
<mml:math id="m77">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and the Dijkstra algorithm in terms of the quantile-quantile (Q-Q) plot between the optimal route costs of the two algorithms for the following list of 10 OD pairs: (1,25), (5,21), (10,16), (6,24), (22,4), (4,16), (18,5), (12,20), (17,5) and (22,9).</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Q-Q plot between the optimal route costs for <inline-formula id="inf77">
<mml:math id="m80">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and Dijkstra over 10 OD pairs.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g009.tif"/>
</fig>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Dynamic network</title>
<p>When traffic is introduced into the simulated network, the network becomes dynamic and link costs are estimated using methods described in <xref ref-type="sec" rid="s3-2">Section 3.2</xref>. Dijkstra, <inline-formula id="inf78">
<mml:math id="m81">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and Q-routing are tested in this dynamic network. Q-routing parameters are found to be optimal with a learning rate of 0.29 and 200 iterations. <xref ref-type="fig" rid="F10">Figure 10</xref> shows a graph with route cost for OD &#x3d;(1,20) for each learning rate. Route cost starts to be minimal for a learning rate of <inline-formula id="inf79">
<mml:math id="m82">
<mml:mrow>
<mml:mn>29</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Route cost and learning rate for Q-routing.</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g010.tif"/>
</fig>
<p>In the dynamic simulated network, <inline-formula id="inf80">
<mml:math id="m83">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, Dijkstra, and Q-routing are invoked every 10&#xa0;s of simulation with all OD pairs, and a few OD pairs are selected for illustrating and comparing route costs. <xref ref-type="fig" rid="F11">Figure 11</xref>,13,12,14 show the cost in terms of sum of the travel times for minimum cost routes for OD &#x3d; <inline-formula id="inf81">
<mml:math id="m84">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1,25</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>5,21</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>4,16</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and (6,14). It can be seen that the data points for Q-routing overlap with those for Dijkstra. Moreover, <xref ref-type="sec" rid="s12">Supplementary Appendix A</xref> illustrates the Q-Q plots between route costs between <inline-formula id="inf82">
<mml:math id="m85">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, Q-routing, and Dijkstra for 10 OD pairs: (1,25), (5,21), (10,16), (6,24), (22,4), (4,16), (18,5), (12,20), (17,5), (22,9).</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Minimum cost route for OD &#x3d; (1,25) with Q-routing, Dijkstra (orange &#x2018;&#x2b;&#x2019;), and <inline-formula id="inf85">
<mml:math id="m88">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> (green&#x2018;x&#x2019;).</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g011.tif"/>
</fig>
<p>Based on <xref ref-type="fig" rid="F11">Figures 11</xref>&#x2013;<xref ref-type="fig" rid="F14">14</xref>, it is clear that the <inline-formula id="inf86">
<mml:math id="m89">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> algorithm produces non-optimal shortest routes, resulting in larger route costs as compared to those achieved by Dijkstra and Q-routing. This is because the heuristic function for <inline-formula id="inf87">
<mml:math id="m90">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">A</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, used in this simulated network no longer remains admissible as the link costs become dynamic. At any node during the <inline-formula id="inf88">
<mml:math id="m91">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> run-time, the heuristic function always overestimates the route cost to the destination.</p>
<fig id="F12" position="float">
<label>FIGURE 12</label>
<caption>
<p>Minimum cost route for OD &#x3d; 5,21 with Q-routing, Dijkstra (orange &#x2018;&#x2b;&#x2019;), and A&#x2a;(green&#x2018;x&#x2019;).</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g012.tif"/>
</fig>
<fig id="F13" position="float">
<label>FIGURE 13</label>
<caption>
<p>Minimum cost route for OD &#x3d; 4,16 with Q-routing, Dijkstra (orange &#x2018;&#x2b;&#x2019;), and A&#x2a;(green &#x2018;x&#x2019;).</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g013.tif"/>
</fig>
<fig id="F14" position="float">
<label>FIGURE 14</label>
<caption>
<p>Minimum cost route for OD &#x3d; 6,24 with Q-routing, Dijkstra (orange &#x2018;&#x2b;&#x2019;), and A&#x2a;(green &#x2018;x&#x2019;).</p>
</caption>
<graphic xlink:href="ffutr-06-1524232-g014.tif"/>
</fig>
<p>On the other hand, Q-routing delivers comparable route costs to Dijkstra during the entire simulation of the network. Following points should be considered while interpreting the plots in <xref ref-type="sec" rid="s12">Supplementary Appendix Figures A1&#x2013;A10</xref> of <xref ref-type="sec" rid="s12">Supplementary Appendix A</xref>. In a perfect information setting wherein the true link costs are accessible to the algorithm, it is well known that the Dijkstra algorithm gives the optimal shortest path between any two nodes of a network (<xref ref-type="bibr" rid="B6">Cormen et al., 2022</xref>). That is, Dijkstra gives the best-case cost when correct link costs are known. The costs corresponding to the Dijkstra algorithm shown in the plots of <xref ref-type="sec" rid="s12">Supplementary Appendix Figures A1&#x2013;A10</xref> are those when Dijkstra was run as if the true costs were known. Although the true costs are not known in a realistic setting, the x-axis coordinates in the aforementioned Q-Q plots represent the theoretical upper bound of the performance for respective OD pairs. Along the y-axis we plot the costs produced by the Q-learning algorithm, wherein Q-learning is performing routing without knowing the costs. Thus, if in the Q-Q plot, the scatter is around the 45&#xb0; line, it goes on to showing that the Q-learning algorithm, under an imperfect information setting, tends to perform close to an algorithm which gives the theoretical upper bound under perfect information setting. This is our main result, i.e., Q-learning tends to give near optimal result. Further, <xref ref-type="sec" rid="s12">Supplementary Appendix B</xref> visualizes the agreement and differences in shortest routes estimated by Q-routing, Dijkstra and <inline-formula id="inf89">
<mml:math id="m92">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>A</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> for OD &#x3d;(1,25) pair.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>In this paper, we have reviewed traditional shortest path finding algorithms, and have investigated RL-based search strategy, Q-routing, in a simulated transportation network. We have presented a comparison of Q-routing algorithms with the performance of two benchmark algorithms, namely, Dijkstra and A&#x2a; in the network with dynamic link costs.</p>
<p>We have demonstrated the feasibility of Q-routing in a network with dynamically changing link-costs. We have shown that Q-routing estimates the cost route as minimal as Dijkstra&#x2019;s and can be considered as an alternative optimal algorithm for finding the shortest path. Despite the time involved while learning the link-costs, during offline execution phase, Q-routing, however, may be executed with run-time complexity as equal as Dijkstra&#x2019;s during online phase or subsequent runs. Also, Q-routing offers more options to expand the number of features used in estimating shortest path with a deep neural network to model an approximation of minimal travel time based on the number of selected features. On the other hand, with the dynamic link costs, the heuristic function of A&#x2a; becomes less effective, producing suboptimal results. In a broader network scenario, a neural network can allow to learn statistics from from multiple network features such as the number of vehicles that enter the segment, segment length, number of lanes, current time of the day, and current weather conditions on the road. The cost function of the deep RL model might be a weighted combination of the output of the neural network and objective rewards (travel time, distance, operation cost, etc.), and the output of the model is still a policy to make the shortest path decision.</p>
<p>In this study, a number of factors may have limited the range of experimentation, i.e., whether it is possible to construct an A&#x2a; guiding heuristic function based on estimated travel time, or it is challenging to simulate a city-scaled transportation networks for a more realistic case study. Also, not all algorithms reviewed in this paper are programmed since it is tedious to implement same-class algorithms that are computationally more expensive.</p>
<p>For future work, deep Q-learning (<xref ref-type="bibr" rid="B38">Van Hasselt et al., 2016</xref>) using a variety of architectures such as Graph Convolutional Networks (<xref ref-type="bibr" rid="B44">Zhang et al., 2019</xref>) and Graph Attention Networks (<xref ref-type="bibr" rid="B39">Veli&#x10d;kovi&#x107; et al., 2018</xref>), can be integrated into the reinforcement learning based Q-routing to estimate shortest paths directly from network features such as traffic counts and link characteristics, rather than using a tabular-based estimation of minimal cost route and the link costs estimated from space-mean speeds as demonstrated in this paper.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>HP: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Project administration, Resources, Software, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. SN: Resources, Software, Writing&#x2013;original draft, Writing&#x2013;review and editing. BM: Conceptualization, Funding acquisition, Methodology, Resources, Writing&#x2013;review and editing. EM: Writing&#x2013;review and editing. AA: Conceptualization, Formal Analysis, Funding acquisition, Investigation, Methodology, Resources, Visualization, Writing&#x2013;review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. The authors thank the UK Research and Innovation (UKRI) and the Natural Sciences and Engineering Research Council of Canada (NSERC) for funding this study (Grant Number: ALLRP 548594-2019).</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/ffutr.2025.1524232/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/ffutr.2025.1524232/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>The use of the descriptor &#x2018;Vehicular&#x2019; in VANETS should not be confused to imply as if they represent transportation or vehicular traffic networks. VANETS are mobile communication networks.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Alam</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Moh</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Survey on Q-learning-based position-aware routing protocols in flying <italic>ad hoc</italic> networks</article-title>. <source>Electronics</source> <volume>11</volume> (<issue>7</issue>), <fpage>1099</fpage>. <pub-id pub-id-type="doi">10.3390/electronics11071099</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Al-Rawi</surname>
<given-names>H. A. A.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Yau</surname>
<given-names>K.-L. A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Application of reinforcement learning to routing in distributed wireless networks: a review</article-title>. <source>Artif. Intell. Rev.</source> <volume>43</volume>(<issue>3</issue>), <fpage>381</fpage>&#x2013;<lpage>416</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-012-9383-6</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhavsar</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Chowdhury</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Rahman</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>A network wide simulation strategy of alternative fuel vehicles</article-title>. <source>Transp. Res. Part C Emerg. Technol.</source> <volume>40</volume>, <fpage>201</fpage>&#x2013;<lpage>214</lpage>. <pub-id pub-id-type="doi">10.1016/j.trc.2013.12.013</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boyan</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Littman</surname>
<given-names>M. L.</given-names>
</name>
</person-group> (<year>1994</year>). <article-title>Packet routing in dynamically changing networks: a reinforcement learning approach</article-title>. <source>Adv. Neural Inf. Process. Syst.</source>, <fpage>671</fpage>&#x2013;<lpage>678</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chin</surname>
<given-names>Y. K.</given-names>
</name>
<name>
<surname>Kow</surname>
<given-names>W. Y.</given-names>
</name>
<name>
<surname>Khong</surname>
<given-names>W. L.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>M. K.</given-names>
</name>
<name>
<surname>Teo</surname>
<given-names>K. T. K.</given-names>
</name>
</person-group> (<year>2012</year>). &#x201c;<article-title>Q-learning traffic signal optimization within multiple intersections traffic network</article-title>,&#x201d; in <source>2012 sixth UKSim/AMSS European symposium on computer modeling and simulation</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>343</fpage>&#x2013;<lpage>348</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cormen</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Leiserson</surname>
<given-names>C. E.</given-names>
</name>
<name>
<surname>Rivest</surname>
<given-names>R. L.</given-names>
</name>
<name>
<surname>Stein</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Introduction to algorithms</source>. <publisher-name>MIT press</publisher-name>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dijkstra</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>1959</year>). <article-title>A note on two problems in connexion with graphs</article-title>. <source>Numer. Math.</source> <volume>1</volume> (<issue>1</issue>), <fpage>269</fpage>&#x2013;<lpage>271</lpage>. <pub-id pub-id-type="doi">10.1007/bf01386390</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ducrocq</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Farhi</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Deep reinforcement q-learning for intelligent traffic signal control with partial detection</article-title>. <source>Int. J. intelligent Transp. Syst. Res.</source> <volume>21</volume> (<issue>1</issue>), <fpage>192</fpage>&#x2013;<lpage>206</lpage>. <pub-id pub-id-type="doi">10.1007/s13177-023-00346-4</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Farazi</surname>
<given-names>N. P.</given-names>
</name>
<name>
<surname>Zou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ahamed</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Barua</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Deep reinforcement learning in transportation research: a review</article-title>. <source>Transp. Res. Interdiscip. Perspect.</source> <volume>11</volume>, <fpage>100425</fpage>. <pub-id pub-id-type="doi">10.1016/j.trip.2021.100425</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fellendorf</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Vortisch</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Microscopic traffic flow simulator vissim</article-title>. <source>Simul. Int. Ser. Operations Res. and Manag. Sci.</source>, <fpage>63</fpage>&#x2013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1007/978-1-4419-6142-6_2</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fua</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rilett</surname>
<given-names>L. R.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Heuristic shortest path algorithms for transportation applications: state of the art</article-title>. <source>Comput. Operations Res.</source> <volume>33</volume>, <fpage>3324</fpage>&#x2013;<lpage>3343</lpage>.</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hancock</surname>
<given-names>P. A.</given-names>
</name>
<name>
<surname>Nourbakhsh</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Stewart</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>On the future of transportation in an era of automated and autonomous vehicles</article-title>. <source>Proc. Natl. Acad. Sci.</source> <volume>116</volume> (<issue>16</issue>), <fpage>7684</fpage>&#x2013;<lpage>7691</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1805770115</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hart</surname>
<given-names>P. E.</given-names>
</name>
<name>
<surname>Nilsson</surname>
<given-names>N. J.</given-names>
</name>
<name>
<surname>Raphael</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>1968</year>). &#x201c;<article-title>A formal basis for the heuristic determination of minimum cost paths</article-title>,&#x201d; in <source>IEEE transactions on systems science and cybernetics</source> (<publisher-name>IEEE</publisher-name>), <fpage>100</fpage>&#x2013;<lpage>107</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hedrick</surname>
<given-names>C. L.</given-names>
</name>
</person-group> (<year>1988</year>). <source>RFC1058: routing information protocol</source>. <publisher-loc>USA</publisher-loc>: <publisher-name>RFC Editor</publisher-name>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang, Q.W.</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhan</surname>
<given-names>F. B.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>A shortest path algorithm with novel heuristics for dynamic transportation networks</article-title>. <source>Int. J. Geogr. Inf. Sci.</source> <volume>21</volume> (<issue>6</issue>), <fpage>625</fpage>&#x2013;<lpage>644</lpage>. <pub-id pub-id-type="doi">10.1080/13658810601079759</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Katre</surname>
<given-names>P. R.</given-names>
</name>
<name>
<surname>Thakare</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>A survey on shortest path algorithm for road network in emergency services</article-title>,&#x201d; in <source>International conference for convergence in Technology (I2CT)</source>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koenig</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Likhachev</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Incremental A</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>14</volume>.</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Koenig</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Likhachev</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2002</year>). &#x201c;<article-title>D&#x2a; lite</article-title>,&#x201d; in <source>Eighteenth national conference on artificial intelligence</source>, <fpage>476</fpage>&#x2013;<lpage>483</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Real-time deep reinforcement learning based vehicle navigation</article-title>. <source>Appl. Soft Comput.</source> <volume>96</volume>, <fpage>106694</fpage>. <pub-id pub-id-type="doi">10.1016/j.asoc.2020.106694</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Qgrid: Q-learning based routing protocol for vehicular <italic>ad hoc</italic> networks</article-title>,&#x201d; in <source>2014 IEEE 33rd international performance computing and communications conference (IPCCC)</source>, <fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/PCCC.2014.7017079</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luong</surname>
<given-names>N. C.</given-names>
</name>
<name>
<surname>Hoang</surname>
<given-names>D. T.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Niyato</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.-C.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Applications of deep reinforcement learning in communications and networking: a survey</article-title>. <source>IEEE Commun. Surv. Tutorials</source> <volume>21</volume> (<issue>4</issue>), <fpage>3133</fpage>&#x2013;<lpage>3174</lpage>. <pub-id pub-id-type="doi">10.1109/COMST.2019.2916583</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Madkour</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Aref</surname>
<given-names>W. G.</given-names>
</name>
<name>
<surname>Rehman</surname>
<given-names>F. U.</given-names>
</name>
<name>
<surname>Rahman</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Basalamah</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>A survey of shortest-path algorithms</article-title>. <comment>arXiv preprint arXiv:1705.02044</comment>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Magzhan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Jani</surname>
<given-names>H. M.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>A review and evaluations of shortest path algorithms</article-title>. <source>Int. J. Sci. Technol. Res.</source> <volume>2</volume> (<issue>6</issue>), <fpage>99</fpage>&#x2013;<lpage>104</lpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mammeri</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Reinforcement learning based routing in networks: review and classification of approaches</article-title>. <source>IEEE Access</source> <volume>7</volume>, <fpage>55916</fpage>&#x2013;<lpage>55950</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2913776</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Meerhof</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Loop-breaking approaches for vehicle route planning with multi-agent q-routing. B.S. thesis</source>. <publisher-name>University of Twente</publisher-name>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moreno-Malo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Posadas-Yag&#xfc;e</surname>
<given-names>J.-L.</given-names>
</name>
<name>
<surname>Cano</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Calafate</surname>
<given-names>C. T.</given-names>
</name>
<name>
<surname>Conejero</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Poza-Lujan</surname>
<given-names>J.-L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Improving traffic light systems using deep q-networks</article-title>. <source>Expert Syst. Appl.</source> <volume>252</volume>, <fpage>124178</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2024.124178</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pallottino</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Scutella</surname>
<given-names>M. G.</given-names>
</name>
</person-group> (<year>1998</year>). &#x201c;<article-title>Shortest path algorithms in transportation models: classical and innovative aspects</article-title>,&#x201d; in <source>Equilibrium and advanced transportation modelling</source>, <fpage>245</fpage>&#x2013;<lpage>281</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Pearl</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>1984</year>). <source>Heuristics: intelligent search strategies for computer problem solving</source>, <fpage>48</fpage>. <publisher-name>Addison-Wesley</publisher-name>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Polydoros</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nalpantidis</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Survey of model-based reinforcement learning: applications on robotics</article-title>. <source>J. Intelligent and Robotic Syst.</source> <volume>86</volume>, <fpage>153</fpage>&#x2013;<lpage>173</lpage>. <pub-id pub-id-type="doi">10.1007/s10846-017-0468-y</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramalingam</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Reps</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1996a</year>). <article-title>On the computational complexity of dynamic graph problems</article-title>. <source>Theor. Comput. Sci.</source> <volume>158</volume> (<issue>1</issue>), <fpage>233</fpage>&#x2013;<lpage>277</lpage>. <pub-id pub-id-type="doi">10.1016/0304-3975(95)00079-8</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramalingam</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Reps</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1996b</year>). <article-title>An incremental algorithm for a generalization of the shortest-path problem</article-title>. <source>J. Algorithms</source> <volume>21</volume> (<issue>2</issue>), <fpage>267</fpage>&#x2013;<lpage>305</lpage>. <pub-id pub-id-type="doi">10.1006/jagm.1996.0046</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rampf</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Grigoropoulos</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Malcolm</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Keler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bogenberger</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Modelling autonomous vehicle interactions with bicycles in traffic simulation</article-title>. <source>Front. Future Transp.</source> <volume>3</volume>, <fpage>894148</fpage>. <pub-id pub-id-type="doi">10.3389/ffutr.2022.894148</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Russell</surname>
<given-names>S. J.</given-names>
</name>
<name>
<surname>Norvig</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2003</year>). <source>Artificial intelligence: a modern approach</source>. <edition>4th edn</edition>. <publisher-name>Pearson</publisher-name>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sunita Kumawat</surname>
<given-names>C. D.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>An extensive review of shortest path problem solving algorithms</article-title>,&#x201d; in <source>Proceedings of the fifth international conference on intelligent computing and control systems</source>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Surekha</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Santosh</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Review of shortest path algorithm</article-title>. <source>Int. Res. J. Eng. Technol. (IRJET)</source> <volume>3</volume> (<issue>8</issue>), <fpage>1956</fpage>&#x2013;<lpage>1959</lpage>.</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sutton</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Barto</surname>
<given-names>A. G.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Reinforcement learning: an introduction</source>. <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Swapno</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nobel</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Meena</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Meena</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Azar</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Haider</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>A reinforcement learning approach for reducing traffic congestion using deep q learning</article-title>. <source>Sci. Rep.</source> <volume>14</volume>, <fpage>30452</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-024-75638-0</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Hasselt</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Guez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Deep reinforcement learning with double Q-learning</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>30</volume>. <pub-id pub-id-type="doi">10.1609/aaai.v30i1.10295</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Veli&#x10d;kovi&#x107;</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cucurull</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Casanova</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lio</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Graph attention networks</article-title>,&#x201d; in <source>International conference on learning representations</source>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vinyals</surname>
<given-names>B. I. C. W. M. e.a. O.</given-names>
</name>
<name>
<surname>Babuschkin</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Czarnecki</surname>
<given-names>W. M.</given-names>
</name>
<name>
<surname>Mathieu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Dudzik</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Grandmaster level in starcraft ii using multi-agent reinforcement learning</article-title>. <source>Nature</source> <volume>575</volume>, <fpage>350</fpage>&#x2013;<lpage>354</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-019-1724-z</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lan</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>An intelligent self-driving truck system for highway transportation</article-title>. <source>Front. Neurorobotics</source> <volume>16</volume>, <fpage>843026</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2022.843026</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhan</surname>
<given-names>F. B.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Three fastest shortest path algorithms on real road networks: data structures and procedures</article-title>. <source>J. Geogr. Inf. Decis. Analysis</source> <volume>1</volume> (<issue>1</issue>), <fpage>70</fpage>&#x2013;<lpage>82</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhan</surname>
<given-names>F. B.</given-names>
</name>
<name>
<surname>Noon</surname>
<given-names>C. E.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Shortest path algorithms: an evaluation using real road networks</article-title>. <source>Transp. Sci.</source> <volume>32</volume> (<issue>1</issue>), <fpage>65</fpage>&#x2013;<lpage>73</lpage>. <pub-id pub-id-type="doi">10.1287/trsc.32.1.65</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tong</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Maciejewski</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Graph convolutional networks: a comprehensive review</article-title>. <source>Comput. Soc. Netw.</source> <volume>6</volume> (<issue>1</issue>), <fpage>11</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1186/s40649-019-0069-y</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>