<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1511712</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>AMaze: an intuitive benchmark generator for fast prototyping of generalizable agents</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Godin-Dubois</surname> <given-names>Kevin</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2852838/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Miras</surname> <given-names>Karine</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/899991/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Kononova</surname> <given-names>Anna V.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Computer Science Department, Vrije Universiteit Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country></aff>
<aff id="aff2"><sup>2</sup><institution>LIACS, Leiden University</institution>, <addr-line>Leiden</addr-line>, <country>Netherlands</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Xin Zhang, Chinese Academy of Sciences (CAS), China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Siao Wang, Xtreme Intelligence LLC, China</p>
<p>Qing Gao, Sun Yat-sen University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Kevin Godin-Dubois <email>k.j.m.godin-dubois&#x00040;vu.nl</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>03</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1511712</elocation-id>
<history>
<date date-type="received">
<day>15</day>
<month>10</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>02</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Godin-Dubois, Miras and Kononova.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Godin-Dubois, Miras and Kononova</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Traditional approaches to training agents have generally involved a single, deterministic environment of minimal complexity to solve various tasks such as robot locomotion or computer vision. However, agents trained in static environments lack generalization capabilities, limiting their potential in broader scenarios. Thus, recent benchmarks frequently rely on multiple environments, for instance, by providing stochastic noise, simple permutations, or altogether different settings. In practice, such collections result mainly from costly human-designed processes or the liberal use of random number generators. In this work, we introduce AMaze, a novel benchmark generator in which embodied agents must navigate a maze by interpreting visual signs of arbitrary complexities and deceptiveness. This generator promotes human interaction through the easy generation of feature-specific mazes and an intuitive understanding of the resulting agents&#x00027; strategies. As a proof-of-concept, we demonstrate the capabilities of the generator in a simple, fully discrete case with limited deceptiveness. Agents were trained under three different regimes (one-shot, scaffolding, and interactive), and the results showed that the latter two cases outperform direct training in terms of generalization capabilities. Indeed, depending on the combination of generalization metric, training regime, and algorithm, the median gain ranged from 50% to 100% and maximal performance was achieved through interactive training, thereby demonstrating the benefits of a controllable human-in-the-loop benchmark generator.</p></abstract>
<kwd-group>
<kwd>benchmark</kwd>
<kwd>human-in-the-loop</kwd>
<kwd>generalization</kwd>
<kwd>mazes</kwd>
<kwd>Reinforcement Learning</kwd>
</kwd-group>
<contract-sponsor id="cn001">Vrije Universiteit Amsterdam<named-content content-type="fundref-id">https://doi.org/10.13039/501100001833</named-content></contract-sponsor>
<counts>
<fig-count count="7"/>
<table-count count="3"/>
<equation-count count="4"/>
<ref-count count="32"/>
<page-count count="11"/>
<word-count count="7042"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Based on the need to fairly compare algorithms (Henderson et al., <xref ref-type="bibr" rid="B13">2018</xref>), benchmarks have proliferated in the Reinforcement Learning (RL) community. These cover a wide range of tasks, from the full collection of Atari 2,600 games (Bellemare et al., <xref ref-type="bibr" rid="B5">2013</xref>) to 3D simulations in Mujoco (Laskin et al., <xref ref-type="bibr" rid="B16">2021</xref>). However, in recent years, the focus of research has changed from producing more complex environments to producing a <italic>range</italic> of environments. Although undeniable progress has been made with respect to the capabilities of trained agents, much remains to be done for their capacity to generalize (Mnih et al., <xref ref-type="bibr" rid="B18">2015</xref>). In practice, agents &#x0201C;will not learn a general policy, but instead a policy that will only work for a particular version of a particular task with particular initial parameters&#x0201D; (Risi and Togelius, <xref ref-type="bibr" rid="B21">2020</xref>).</p>
<p>Thus, a recurring theme in modern RL research is the training of agents in various situations to avoid overfitting. Although some algorithms have built-in solutions to smooth out the learning process, e.g., TD3 (Fujimoto et al., <xref ref-type="bibr" rid="B9">2018</xref>) (where small perturbations are applied to the actions), providing such a diversity of experience primarily originates from the environments themselves. To this end, numerous benchmarks now consist of a collection with varying degrees of homogeneity. Some of them have a similar structure, as in the Sonic benchmark (Nichol et al., <xref ref-type="bibr" rid="B19">2018</xref>) where levels are small areas taken from three games in the franchise. In other cases, environments share very little: In the Arcade Learning Environment (ALE), the single common factors are the dimensions of the observation space (Bellemare et al., <xref ref-type="bibr" rid="B5">2013</xref>). Intermediate test suites with distinct but complementary sets of &#x0201C;skill-building&#x0201D; tasks have also been designed, for example, with the Mujoco simulator (Wawrzy&#x00144;ski, <xref ref-type="bibr" rid="B31">2009</xref>; Yu et al., <xref ref-type="bibr" rid="B32">2019</xref>; Laskin et al., <xref ref-type="bibr" rid="B16">2021</xref>) or Meta-World (Yu et al., <xref ref-type="bibr" rid="B32">2019</xref>). However, all of these examples share a common feature: the set of environments is predefined, generally the result of a costly human-tailored design procedure, e.g., Beattie et al. (<xref ref-type="bibr" rid="B4">2016</xref>).</p>
<p>To solve this generalization problem, agents must face sufficiently diverse situations so that the underlying principles are learned instead of a specific trajectory. Naturally, this requires generating environments that exhibit such diversity while still offering the same core challenges. For instance, in the context of maze navigation, reaching e.g., the exit might be the goal while the actual topology of said maze is only relevant insofar as giving agents a wide sample of states to learn from. One common way to address this later point is to use procedural generation (Beattie et al., <xref ref-type="bibr" rid="B4">2016</xref>; Kempka et al., <xref ref-type="bibr" rid="B15">2016</xref>; Harries et al., <xref ref-type="bibr" rid="B12">2019</xref>; Juliani et al., <xref ref-type="bibr" rid="B14">2019</xref>; Tomilin et al., <xref ref-type="bibr" rid="B28">2022</xref>) or complementary techniques such as evolutionary algorithms (Alaguna and Gomez, <xref ref-type="bibr" rid="B1">2018</xref>; Wang et al., <xref ref-type="bibr" rid="B30">2019</xref>). For example, ProcGen (Cobbe et al., <xref ref-type="bibr" rid="B7">2020</xref>) encompasses 16 different types of environment and serves as a generalizable alternative to ALE. Adapting more recent video game environments, either directly (Synnaeve et al., <xref ref-type="bibr" rid="B25">2016</xref>) or in a light format (Tian et al., <xref ref-type="bibr" rid="B26">2017</xref>), can help further push adaptability by providing finer-grained perceptions and actions. While such an approach can be used to create large training sets, the main difficulty becomes the design of a sufficiently tunable generator, i.e., one in which desirable features are easy to introduce.</p>
<p>Considering the challenges of generating a panel of demanding training environments, the contribution of this article is two-fold:</p>
<list list-type="order">
<list-item><p>We introduce AMaze,<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> a generator for generic, computationally inexpensive environments of unbounded complexity that focus on generalization (via environmental diversity) and intelligibility (intuitive human understanding).</p></list-item>
<list-item><p>We demonstrate how such a generator is helpful in leading to more generalized performance (robust behavior w.r.t. unseen tasks) and how it can benefit from human input (e.g., to dynamically adjusting difficulty).</p></list-item>
</list>
<p>After highlighting, in <xref ref-type="sec" rid="s2">Section 2</xref>, the niche this generator occupies in the current benchmark literature, we describe its main components in <xref ref-type="sec" rid="s3">Section 3</xref>. Three alternative methodologies for training generalized maze-navigating agents are then detailed in <xref ref-type="sec" rid="s4">Section 4</xref> alongside two algorithms (A2C and PPO). The resulting performance in handling unknown environments is then thoroughly tested in <xref ref-type="sec" rid="s5">Section 5</xref>, allowing us to draw conclusions about the relative benefits of the generator, the training processes, and the underlying algorithms.</p></sec>
<sec id="s2">
<title>2 Related benchmarks</title>
<p>To place this generator in perspective, we conducted an extensive comparison with a select number of commonly used benchmarks. As our library is primarily targeted at Python environments, we restricted the set of considered environments to those that could be reliably installed and used on an experimenter&#x00027;s machine. Timing was done on 1000 time steps averaged over 10 replicates on an i7-1185G7 (3GHz) using the Python 3.10 version of all libraries, except for the Unsupervised Reinforcement Learning Benchmark (URLB) (Laskin et al., <xref ref-type="bibr" rid="B16">2021</xref>) and RetroGym (Nichol et al., <xref ref-type="bibr" rid="B19">2018</xref>) which required Python 3.8. In the latter case, we used the ROMs linked in the library&#x00027;s documentation. The scripts, intermediate data and figures are available as part of AMaze&#x00027;s repository.</p>
<p>As detailed in the following section, AMaze can provide environments for fully discrete, fully continuous, and hybrid agents. <xref ref-type="table" rid="T1">Table 1</xref> illustrates how the former case allows for fast simulation at the cost of low observable complexity. Based on the time taken to simulate 1,000 timesteps, only the simplest of the gymnasium suite (Sutton and Barto, <xref ref-type="bibr" rid="B24">2018</xref>) is comparable to AMaze which, in addition, provides numerous unique and experimenter-controlled environments. In the hybrid case, where agents perceive images but still only take discrete steps, the library is on par with Classic Control tasks (Barto et al., <xref ref-type="bibr" rid="B2">1983</xref>) such as Mountain Car or Cart Pole. ProcGen (Cobbe et al., <xref ref-type="bibr" rid="B7">2020</xref>) addresses similar concerns as AMaze and is quite comparable in terms of speed, but has a stronger focus on randomness, with difficulty levels being the main way of controlling the resulting environments. DeepMind Lab2D (Beattie et al., <xref ref-type="bibr" rid="B3">2020</xref>), while noticably slower, is also extensively customizable, albeit through lua scripting, and allows for heterogeneous multi-agent experiments.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparison between AMaze and related benchmark (suites). All time metrics (performance) correspond to the wall time for 1,000 timesteps of the corresponding environment, averaged over 10 replicates. Qualification of the inputs, outputs, and control levels are taken from the related article or, when unavailable, directly from the sources. Overall, AMaze is competitive with small-scale benchmarks, but provides the experimenter with more control over the characteristics of the targeted environments. Complex environments (e.g., 3D) have much higher computational costs making AMaze an efficient and scalable prototyping platform.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th/>
<th/>
<th/>
<th/>
<th/>
<th valign="top" align="center" colspan="2"><bold>Time (s)</bold></th>
</tr>
<tr>
<th valign="top" align="center"><bold>Library</bold></th>
<th valign="top" align="center"><bold>Family</bold></th>
<th valign="top" align="center"><bold>N</bold></th>
<th valign="top" align="center"><bold>Inputs</bold></th>
<th valign="top" align="center"><bold>Outputs</bold></th>
<th valign="top" align="center"><bold>Control</bold></th>
<th valign="top" align="center"><bold>Median</bold></th>
<th valign="top" align="center"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-i0001.tif"/></th>
</tr>
</thead>
<tbody>
<tr>
<td/>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">Extensive</td>
<td valign="top" align="center">0.010</td>
</tr>
 <tr>
<td valign="top" align="left">A Maze</td>
<td valign="top" align="center">Hybrid</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">Image</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">Extensive</td>
<td valign="top" align="center">0.025</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">64</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">Extensive</td>
<td valign="top" align="center">0.102</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Toy text</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">None</td>
<td valign="top" align="center">0.009</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Classic control</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">Both</td>
<td valign="top" align="center">None</td>
<td valign="top" align="center">0.023</td>
</tr>
 <tr>
<td valign="top" align="left">Gymnasium</td>
<td valign="top" align="center">Mujoco</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">None</td>
<td valign="top" align="center">0.090</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Box2D</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">Both</td>
<td valign="top" align="center">None</td>
<td valign="top" align="center">0.112</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">ALE</td>
<td valign="top" align="center">104</td>
<td valign="top" align="center">Image</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">Modes</td>
<td valign="top" align="center">0.400</td>
</tr>
 <tr>
<td valign="top" align="left">Miscellaneous</td>
<td valign="top" align="center">ProcGen</td>
<td valign="top" align="center">100</td>
<td valign="top" align="center">Image</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">Modes</td>
<td valign="top" align="center">0.031</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Lab2D</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">Both</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">Script</td>
<td valign="top" align="center">0.056</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">MetaWorld</td>
<td valign="top" align="center">50</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">None</td>
<td valign="top" align="center">1.228</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">URLB</td>
<td valign="top" align="center">24</td>
<td valign="top" align="center">Both</td>
<td valign="top" align="center">Continuous</td>
<td valign="top" align="center">None</td>
<td valign="top" align="center">3.929</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">GameBoy</td>
<td valign="top" align="center">27</td>
<td/>
<td/>
<td/>
<td valign="top" align="center">0.197</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Sms</td>
<td valign="top" align="center">88</td>
<td/>
<td/>
<td/>
<td valign="top" align="center">0.269</td>
</tr>
 <tr>
<td valign="top" align="left">Retro-Gym</td>
<td valign="top" align="center">Nes</td>
<td valign="top" align="center">295</td>
<td valign="top" align="center">Image</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">None</td>
<td valign="top" align="center">0.355</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Genesis</td>
<td valign="top" align="center">331</td>
<td/>
<td/>
<td/>
<td valign="top" align="center">0.472</td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Snes</td>
<td valign="top" align="center">184</td>
<td/>
<td/>
<td/>
<td valign="top" align="center">0.736</td>
</tr>
 <tr>
<td valign="top" align="left">VizDoom</td>
<td valign="top" align="center">MazeExplorer</td>
<td valign="top" align="center">81</td>
<td valign="top" align="center">Image</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">Extensive</td>
<td valign="top" align="center">0.553</td>
</tr>
<tr>
<td/>
<td valign="top" align="center">LevDoom</td>
<td valign="top" align="center">72</td>
<td valign="top" align="center">Image</td>
<td valign="top" align="center">Discrete</td>
<td valign="top" align="center">None</td>
<td valign="top" align="center">1.059</td>
</tr></tbody>
</table>
</table-wrap>
<p>With respect to the fully continuous case, the most computationally expensive of the three regimes, AMaze performs at a level similar to that of Box2D or Mujoco (Todorov et al., <xref ref-type="bibr" rid="B27">2012</xref>), which, in traditional implementations such as gymnasium (Towers et al., <xref ref-type="bibr" rid="B29">2024</xref>), lack customization capabilities. Purely vision-based benchmarks such as ALE (Bellemare et al., <xref ref-type="bibr" rid="B5">2013</xref>), RetroGym (Nichol et al., <xref ref-type="bibr" rid="B19">2018</xref>), Meta-world (Yu et al., <xref ref-type="bibr" rid="B32">2019</xref>), Unsupervised Reinforcement Learning Benchmark (Laskin et al., <xref ref-type="bibr" rid="B16">2021</xref>), LevDoom (Tomilin et al., <xref ref-type="bibr" rid="B28">2022</xref>), or Maze Explorer (Harries et al., <xref ref-type="bibr" rid="B12">2019</xref>), while offering a more challenging task than AMaze, also exhibit drastically higher costs with variable levels of experimenter control over the environments.</p>
<p>To summarize, AMaze has comparable computational costs with the most simple of environments (e.g., Classic Control or Toy Text) while providing much finer grained control over the challenges proposed to the agent. Conversely, even in its most expensive variation (fully continuous) it outperforms complex settings such as Mujoco-based environments, ALE or Retro-Gym. Combined with its extensive and intuitive parameterization, this makes it an ideal platform for trying out new algorithms, policies or hypotheses before deployment on more demanding contexts.</p>
<p>It follows that AMaze fills a very specific niche in the benchmarking landscape by providing a computationally inexpensive framework to design challenging environments. Control over the various characteristics of said environments is left in the hands of the experimenter through a number of high- and low-level parameters that will be described in the following section.</p></sec>
<sec id="s3">
<title>3 Generating mazes</title>
<p>Learning to navigate mazes represents a flexible, diverse, yet challenging tested for training agents. Here, we propose a <italic>generator</italic> (AMaze) for this task with the following primary characteristics:</p>
<list list-type="simple">
<list-item><p><bold>Loose embodiment</bold></p>
<p>The agent has access only to local spatial information (its current cell) and limited temporal information (previous cell). Arbitrarily complex visual-like information is provided to the agent in either discrete (preprocessed) or continuous (image) form as detailed in <xref ref-type="sec" rid="s3.2">Section 3.2</xref>.</p>
</list-item>
<list-item><p><bold>Computational lightweightness</bold></p>
<p>No physics engine or off-screen renderings are required for such 2D mazes. Thus, challenging environments can be generated that are both observably complex (Beattie et al., <xref ref-type="bibr" rid="B3">2020</xref>) and relatively fast (as seen in <xref ref-type="table" rid="T1">Table 1</xref>).</p>
</list-item>
<list-item><p><bold>Open-endedness</bold></p>
<p>As illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>, a given maze results from the interaction of numerous variables controlled by the experimenter, such as its dimensions or the frequency and type of visual cues. In practice, an experimenter can inject any level of complexity into the maze by selecting images of appropriate deceptiveness as cues.</p></list-item>
</list>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Generic maze example. Agents start in one corner and must reach the opposite. Corridors can be empty or contain easily identifiable misleading signs (lures). Signs placed on intersections maybe trustworthy or not depending on whether they are a clue or a trap, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-g0001.tif"/>
</fig>
<p>These features make it possible to generate a wide range of mazes for a variety of purposes, from a fast prototyping RL platform to a testbed for embodied computer vision in indirectly encoded NeuroEvolution. In the remainder of the section, we detail the major components of this generator namely the environment&#x00027;s parameters, the agents&#x00027; capabilities, the reward function and, finally, unifying metrics for comparing widely different mazes.</p>
<sec>
<title>3.1 Maze generation</title>
<p>A maze is defined, at its core, by its size (width, height) and the seed of a random number generator. A depth-first search algorithm is then used to create the various paths and intersections with the arbitrary constraint that the final cell is always diagonally opposed to the starting point (itself a parameter). Additionally, mazes can be made unicursive by blocking every intersection that does not lead directly to the goal. Such mazes are called <italic>trivial</italic>, as an optimal strategy simply requires going forward without hitting any wall. In contrast, general-purpose mazes do have intersections, the correct direction being indicated by a sign, hereafter called a <italic>clue</italic>. This corresponds to the class of <italic>simple</italic> mazes, since making the appropriate move in such cases is entirely context dependent. Each intersection <italic>on the path to the goal</italic> is labeled with such a clue.</p>
<p>However, to provide a sufficient level of difficulty, additional types of sign can also be added to a given maze with a user-defined probability. <italic>Lures</italic>, occurring with probability <italic>p</italic><sub><italic>l</italic></sub>, are easily identifiable erroneous signs that request an immediately unfavorable move (going backward or into a wall). They can be placed on any non-intersectional cell along the path to the solution. <italic>Traps</italic>, replace an existing clue (with probability <italic>p</italic><sub><italic>t</italic></sub>) and instead point to a dead end. These types of sign are much harder to detect as they do not violate local assumptions and can result in large, delayed negative rewards. Mazes containing either of these misleading signs are named accordingly, while mazes containing <italic>both</italic> are called <italic>complex</italic>.</p>
</sec>
<sec id="s3.2">
<title>3.2 Agents and state spaces</title>
<p>To successfully navigate a maze, the learning agent must only rely on the visual contents of its current cell to choose its next action. The framework accounts for three combinations of input/output types: fully discrete, fully continuous, and hybrid (continuous observations with discrete actions). Observations in continuous space imply that cells are perceived directly as images albeit with a lower resolution than that presented to humans. Thus, wall detection may not be initially trivial (even for unicursive mazes), and sign recognition comes into play with the possibility of using different symbols for different sign types. In the discrete space, the agent is fed a sequence of eight floats, corresponding to preprocessed information in direct order (<italic>W</italic><sub><italic>e</italic></sub>, <italic>W</italic><sub><italic>n</italic></sub>, <italic>W</italic><sub><italic>w</italic></sub>, <italic>W</italic><sub><italic>s</italic></sub>, <italic>S</italic><sub><italic>e</italic></sub>, <italic>S</italic><sub><italic>n</italic></sub>, <italic>S</italic><sub><italic>w</italic></sub>, <italic>S</italic><sub><italic>s</italic></sub>), as illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Discrete observation space. <bold>(a)</bold> Visual inputs: <italic>W</italic><sub>&#x0002A;</sub> denotes whether there is a wall in the corresponding direction, as well as the direction of the previous cell; <italic>S</italic><sub>&#x0002A;</sub> is non-zero if a sign points toward the corresponding direction. <bold>(b)</bold> Examples: Sample inputs from cells highlighted in <xref ref-type="fig" rid="F1">Figure 1</xref>, as would be perceived by agents (without geometric relationship).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-g0002.tif"/>
</fig>
<p>In this case, the observations take the form of a monodimensional array containing all eight fields, in direct order. Signs can be differentiated through their associated decimal value, which is fully configurable by the experimenter. In the subsequent experiment, we used a single sign of each type with values of 1.0, 0.5, and 0.25 for clues, traps, and lures, respectively. The walls and the originating direction (limited temporal information) are assigned fixed values of 1.0 and 0.5, respectively.</p>
<p>With respect to actions, a discrete space implies that the agent moves directly from one cell to another by choosing one of the four cardinal directions. In contrast, in a continuous action space, the agent controls only its acceleration.</p>
</sec>
<sec>
<title>3.3 Reward function</title>
<p>An optimal strategy, in the fully discrete case, is one where the agent makes no error: no wall collision, no backward steps, and naturally, correct choices at all intersections. Although identical in the hybrid case, as the increase in observation complexity does not change the fact that there exists only one optimal trajectory, this statement no longer holds for the fully continuous case, at least not in the trivial sense. In fact, by controlling its acceleration, an agent can take shorter paths along corners or even take risks based on assumed corridor lengths.</p>
<p>However, in all cases, the same reward function is used to improve strategies as defined by:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="right"><mml:mtr><mml:mtd><mml:mi>r</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>,</mml:mo><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext class="textrm" mathvariant="normal">if&#x000A0;</mml:mtext><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup><mml:mtext class="textrm" mathvariant="normal">&#x000A0;is the goal</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext class="textrm" mathvariant="normal">if&#x000A0;</mml:mtext><mml:mi>a</mml:mi><mml:mtext class="textrm" mathvariant="normal">&#x000A0;caused a collision</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>b</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext class="textrm" mathvariant="normal">if&#x000A0;</mml:mtext><mml:mi>a</mml:mi><mml:mtext class="textrm" mathvariant="normal">&#x000A0;caused a backward step</mml:mtext></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd><mml:mtd><mml:mtext class="textrm" mathvariant="normal">constant time penalty</mml:mtext></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Given <italic>l</italic>, the length of the optimal trajectory, we define two versions of the reward function: <italic>r</italic> and its normalized version <inline-formula><mml:math id="M2"><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula>. The first is used during the training process to provide large incentive toward reaching the goal, while the second&#x00027;s purpose is to compare performance on mazes with different sizes. Furthermore, we refer to the cumulative (episodic) reward as <italic>R</italic> and <inline-formula><mml:math id="M3"><mml:mover accent="true"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula>, respectively. <xref ref-type="table" rid="T2">Table 2</xref> details the specific values used in this experiment.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Elementary rewards for both versions of the reward function (<xref ref-type="disp-formula" rid="E1">Equation 1</xref>): <italic>r</italic> promotes reaching the goal with a large associated reward, while <inline-formula><mml:math id="M7"><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula> always indicates an optimal strategy with a cumulative reward <inline-formula><mml:math id="M8"><mml:mover accent="true"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula>. <italic>l</italic> is the number of cells on the optimal path between start and finish.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>&#x003C1;<sub><italic>e</italic></sub></bold></th>
<th valign="top" align="center"><bold>&#x003C1;<sub><italic>w</italic></sub></bold></th>
<th valign="top" align="center"><bold>&#x003C1;<sub><italic>b</italic></sub></bold></th>
<th valign="top" align="center"><bold>&#x003C1;<sub><italic>t</italic></sub></bold></th>
<th valign="top" align="center"><bold>Cumulative</bold></th>
</tr>
<tr>
<td valign="top" align="left"><italic>r</italic></td>
<td valign="top" align="center">2<italic>l</italic>&#x02212;1</td>
<td valign="top" align="center">&#x02212;0.1</td>
<td valign="top" align="center">&#x02212;0.2</td>
<td valign="top" align="center">&#x02212;1</td>
<td valign="top" align="center"><italic>R</italic> &#x0003D; <italic>l</italic></td>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M9"><mml:mover accent="true"><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula></td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">&#x02212;0.01</td>
<td valign="top" align="center">&#x02212;0.02</td>
<td valign="top" align="center">&#x02212;1/(<italic>l</italic>&#x02212;1)</td>
<td valign="top" align="center"><inline-formula><mml:math id="M10"><mml:mover accent="true"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:math></inline-formula></td>
</tr></tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>3.4 Evaluating maze complexity</title>
<p>Due to the randomness of the generation process, two mazes with different seeds can have very different characteristics. Thus, to provide a common ground from which mazes can be compared, we define two metrics based on Shannon&#x00027;s entropy (Shannon, <xref ref-type="bibr" rid="B23">1948</xref>). First the <italic>Surprisingness</italic> <italic>S</italic>(<italic>M</italic>) of a maze <italic>M</italic>:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>S</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo>-</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002A;</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>p</italic>(<italic>i</italic>) is the observed frequency of input <italic>i</italic> and <italic>I</italic><sub><italic>M</italic></sub> is the set of inputs encountered when performing an optimal trajectory in <italic>M</italic>. Second, the <italic>Deceptiveness</italic> <italic>D</italic>(<italic>M</italic>) defined as:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">cells</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>:</mml:mo><mml:mn>3</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mo>&#x02200;</mml:mo><mml:mi>c</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">traps</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mo>,</mml:mo><mml:mo>&#x02200;</mml:mo><mml:mi>c</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>M</mml:mi><mml:mo>/</mml:mo><mml:mtext class="textrm" mathvariant="normal">cost</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>c</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0003E;</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x02003;</mml:mtext><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mtext class="textrm" mathvariant="normal">cells</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mi>s</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mtext class="textrm" mathvariant="normal">traps</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>s</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>:</mml:mo><mml:mn>3</mml:mn></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>c</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:munder></mml:mstyle><mml:mo>-</mml:mo><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>|</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>s</mml:mi><mml:mo>|</mml:mo><mml:mi>c</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the cost of <italic>c</italic> is above zero for cells containing traps and lures.</p>
<p>As illustrated in <xref ref-type="fig" rid="F3">Figure 3</xref>, both metrics cover different regions of the maze space. Surprisingness describes the likelihood of encountering numerous infrequent states while traversing the maze. Conversely, the Deceptiveness focuses on the frequency with which deceptive states may be encountered, that is, it captures how &#x0201C;dangerous&#x0201D; the maze is. With respect to other types of tasks, be it in simulation or the physical world, <italic>S</italic>(<italic>M</italic>) accounts for the intrinsic variability of the environment while <italic>D</italic>(<italic>M</italic>) would correspond to the ambiguity of said environment e.g., how frequently similar observations can give divergent results. One can see that, by taking advantage of both types of deceptive signs, Complex mazes exhibit the highest combined difficulty and frequency. Furthermore, even with the limitations of discrete inputs, we can here see how it is theoretically possible to generate mazes of arbitrarily high Surprisingness and Deceptiveness. Additional information and the data set on which these analyses are based can be found in the associated Zenodo record (Godin-Dubois, <xref ref-type="bibr" rid="B10">2024</xref>).</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Distribution of Surprisingness vs. Deceptiveness across 500,000 unique mazes from five different classes. The marginal densities for Surprisingness highlight the low number of different Trivial mazes ([2, 4] range), while classes of increasing difficulty allow for more variations. Examples of outlier mazes from the four main classes are depicted in the borders to illustrate the underlying Surprisingness <bold>(right column)</bold> or lack thereof <bold>(left column)</bold>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-g0003.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4 Training protocol on AMaze</title>
<p>To teach agents generalizable navigation skills, we define a <italic>training</italic> maze, presented in <xref ref-type="fig" rid="F4">Figure 4a</xref>. Although, for simplicity, we only depict one variation of this maze, in practice, the agent is trained on all four rotations (<xref ref-type="fig" rid="F4">Figure 4c</xref>). Thus, the agent will not overfit to a particular upper-diagonal type of behavior, but instead will have to develop a context-dependent strategy. Furthermore, intermediate evaluations of the agent&#x00027;s performance are performed in a similar maze (in terms of complexity) with a different seed, as shown in <xref ref-type="fig" rid="F4">Figure 4b</xref>.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Mazes used in direct training. <bold>(a)</bold> Training: maze used to collect experiences and learn from. <bold>(b)</bold> Evaluation: maze used to periodically evaluate performance. Note that, in practice, the agent experiences mazes as in <bold>(c)</bold>, i.e., with all rotations for both training and evaluations.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-g0004.tif"/>
</fig>
<p>To showcase this benchmark&#x00027;s integration with current Reinforcement Learning frameworks, we used Stable Baselines 3 (Raffin et al., <xref ref-type="bibr" rid="B20">2021</xref>) and more specifically their off-the-shelf Advantage Actor-Critic (A2C) and Proximal Policy Optimization (PPO) algorithms with all hyperparameters kept to their default values (Mnih et al., <xref ref-type="bibr" rid="B17">2016</xref>; Schulman et al., <xref ref-type="bibr" rid="B22">2017</xref>). The total training budget is of 3,000,000 timesteps, divided over the four rotational variations of the training maze with possible early stopping if the optimal trajectory is observed on all evaluation mazes.</p>
<sec>
<title>4.1 Scaffolding</title>
<p>In addition to this <italic>direct</italic> training in a hard maze, we also followed two incremental protocols: an <italic>interpolation</italic> training, which &#x0201C;smoothly&#x0201D; transitions from simple to more complex mazes, and the <italic>EDHuCAT</italic> training, which leverages human creativity and reactivity (Eiben and Smith, <xref ref-type="bibr" rid="B8">2015</xref>). In the former case, agents start from trivial environments and gradually move onto harder challenges. However, the final mazes on which agents are trained and evaluated are identical to those of the direct case.</p>
<p>Succinctly, every atomic parameter is interpolated between the initial and final mazes&#x00027; values according to specific per-field rules, e.g. for the apparition of intersections or traps. In this work, the initial maze is unicursive (no intersections) of size 5 &#x000D7; 5 and eight intermediates are inferred through interpolation. As such a total of ten training stages are performed in this protocol, that is 300,000 timesteps each. In case of early convergence, the remainder of the budget is transferred equally to future stages. For more details, the full spectrum of mazes, in image and textual forms, can be found in <xref ref-type="supplementary-material" rid="SM1">Supplementary material 1</xref>.</p>
</sec>
<sec>
<title>4.2 Interactive training</title>
<p>In the interactive setup, we use the Environment-Driven Human-Controlled Automated Training (EDHuCAT) algorithm, loosely inspired by the EDEnS<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> algorithm. As summarized in <xref ref-type="fig" rid="F8">Algorithm 1</xref>, EDHuCAT operates under the joint principles of concurrency (multiple agents evaluated in parallel) and diversity (multiple mazes are generated by the user/experimenter). The advantage of this method over the simple interpolation between initial and final mazes is that it can take advantage of unforeseen developments that occur in the middle of the training. For instance, if the human agent detects that the learning agent has too much difficulty with some newly presented features, they can decide to decrease the difficulty, select from a wider diversity of mazes, or even increase the difficulty. At the same time, the human component makes it harder for the training algorithm itself (A2C or PPO) due to the potential introduction of so-called moving targets (see Section 5.4). That is, a Human may not follow a strict policy for choosing mazes or agents, whether between replicates or even during a given run. The total budget is the same as for the other protocols; however, as three concurrent evaluations are performed for each stage, an agent in a given stage is only trained for a maximum of 100,000 time steps.</p>
<fig id="F8" position="float">
<label>Algorithm 1</label>
<caption><p>EDHuCAT Algorithm. A human agent is used to perform the <italic>select</italic> and <italic>generate</italic> operations. In this work <italic>K</italic> &#x0003D; 3 and<italic>S</italic> &#x0003D; 10.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-g0008.tif"/>
</fig>
</sec>
</sec>
<sec id="s5">
<title>5 Evaluation of generalized performance</title>
<p>Following the training protocols defined in the previous section, we evaluated the final agents on two complementary tasks to determine whether they had acquired generalized behavior in the target maze class. The first is straightforward: can the agent solve any maze of a given complexity or lower? To answer this, we generated 18 mazes, as shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, based on varying amounts of features (clues, lures, and traps) and Surprisingness. The agents are then evaluated with respect to two goals: their success (do they reach the goal) and their reward (cumulative normalized reward <inline-formula><mml:math id="M11"><mml:mover accent="true"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula>, as in <xref ref-type="disp-formula" rid="E1">Equation 1</xref>). Although this allows for comparison between agents based on performance under &#x0201C;normal conditions,&#x0201D; this method suffers from cumulative failure: an error at a given time point may preclude any further success. In fact, agents who take a wrong turn somewhere have little information on how to get back on track. Thus, sub-optimal strategies may end up indistinguishable from trivially bad ones.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Mazes used for generalization evaluation. The first three columns correspond to different maze classes, while the last three all include traps but with different frequencies (1, 3, 16). Each row corresponds to the minimal, median, and maximal complexity of mazes obtained from a random sample of size 10,000.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-g0005.tif"/>
</fig>
<p>To counteract this trend, we also perform a complementary evaluation in a more abstract context. Because the input is discrete and thus enumerable, we can generate the complete set of possible input arrays. As we know which is the correct decision, we can assess which inputs are correctly processed by the agents among the four classes: <italic>empty</italic> corridor, corridor with <italic>lure</italic>, intersection with <italic>clue</italic>, and intersection with <italic>trap</italic>. Although less &#x0201C;natural,&#x0201D; this method ensures complete coverage of all the possible situations that an agent may encounter on an infinite number of mazes. Conversely, it also implies that we may be testing an agent on input configurations that it has never seen during training.</p>
<p>The interested reader might also refer to <xref ref-type="supplementary-material" rid="SM1">Supplementary material 2</xref> for details on the dynamics of each group&#x00027;s error with the different sign types. The full data for every training dynamics is available in Godin-Dubois (<xref ref-type="bibr" rid="B10">2024</xref>).</p>
<sec>
<title>5.1 Generalized maze-navigation</title>
<p>As summarized in <xref ref-type="fig" rid="F6">Figure 6</xref>, the average cumulative rewards (<inline-formula><mml:math id="M12"><mml:mover accent="true"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>&#x00304;</mml:mo></mml:mover></mml:math></inline-formula>) and the success rates (fraction of mazes whose target was reached) are uniquely distributed according to the training regimen and algorithm. For rewards, both direct and interpolation training have similar trends when using the A2C algorithm (in [&#x02212;2, &#x02212;4]), while EDHuCAT stands out with a more dispersed distribution. When considering the PPO algorithm, there is a clear negative impact of direct training vs. both alternatives. Although EDHuCAT still presents a higher variance than interpolation training, both generated agents who obtained better rewards. Furthermore, in the latter case, PPO significantly outperforms A2C with a <italic>p</italic>-value &#x0003C; 0.0001 for an independent t-test with Benjamini-Hochberg correction (Benjamini and Hochberg, <xref ref-type="bibr" rid="B6">1995</xref>).</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Normalized rewards and maze completion rates across trainers and algorithms. <bold>(a)</bold> Average normalized reward: EDHuCAT is better than direct training, but more dispersed than interpolation. <bold>(b)</bold> Average success rate: PPO is dramatically better for interpolation, while its advantage with EDHuCAT is unclear. Statistical differences were obtained with an independent t test and a Benjamini-Hochberg correction. &#x0002A;: <italic>p</italic>-value &#x0003C; 0.05, &#x0002A;&#x0002A;&#x0002A;: <italic>p</italic>-value &#x0003C; 0.0001.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-g0006.tif"/>
</fig>
<p>This difference is more clearly visible with the maze completion rate (<xref ref-type="fig" rid="F6">Figure 6b</xref>), especially for agents generated by interpolation training: the best A2C agent is comparable to the worst PPO agent. Again, this is strongly confirmed statistically using the same methodology and with a similar p-value. Additionally, it would seem, from this distribution, that A2C is a slightly better choice in static environments (direct/A2C is marginally better than interpolation/A2C) and conversely for dynamical environments (direct/PPO generally lower than interpolation/PPO). As previously, the human interventions promoted by EDHuCAT do not appear to be beneficial to the PPO algorithm.</p>
</sec>
<sec>
<title>5.2 Generalized input-processing</title>
<p>We can make similar observations for the direct input processing test, as illustrated in <xref ref-type="fig" rid="F7">Figure 7</xref>. Selecting the correct action is almost perfectly done by all agents, across all treatments, for the simplest cases (empty corridors and corridors with lures). Surprisingly, the reaction to the presence of a nontrivial sign is handled differently depending on the algorithm. Although PPO seems to be more efficient in detecting clues, A2C shows a better response to traps (<xref ref-type="fig" rid="F7">Figure 7a</xref>). Nonetheless, we can see that, on average, PPO shows clear benefits over A2C (<xref ref-type="fig" rid="F7">Figure 7b</xref>). With this test, we can confirm the advantage of using the former over the latter when facing dynamic environments. The statistical significance is lower than 10<sup>&#x02212;3</sup> and 10<sup>&#x02212;2</sup> for the interpolation training and EDHuCAT, respectively. In contrast, there is a marginally significant negative trend between A2C use and environmental variability.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Correct input processing rate across trainers and algorithms. <bold>(a)</bold> Per sign type: Corridors (with and without lures) are trivial, while A2C detects traps more efficiently than PPO and the other way around. <bold>(b)</bold> Average performance: PPO outperforms A2C except for direct training, while non-stationary training seems overall beneficial. Statistical differences were also obtained with an independent t-test and a Benjamini-Hochberg correction: 0.05 &#x02264; &#x0002A; &#x0003C; 10<sup>&#x02212;2</sup> &#x02264; &#x0002A;&#x0002A; &#x0003C; 10<sup>&#x02212;3</sup> &#x02264; &#x0002A;&#x0002A;&#x0002A; &#x0003C; 10<sup>&#x02212;4</sup>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1511712-g0007.tif"/>
</fig>
</sec>
<sec>
<title>5.3 Aggregated performance</title>
<p>To better compare the general performance of all training regimens and algorithms, we provide the maximum and median performance of the six combinations for the three metrics in <xref ref-type="table" rid="T3">Table 3</xref>. EDHuCAT succeeded in generating the most general maze navigator of all treatments with an average normalized reward of &#x02013;0.498, compared to &#x02013;0.873 and &#x02013;1.2 of direct and interpolation trainings, respectively. Surprisingly, such rewards were obtained with both algorithms, while alternatives fared much worse when using A2C. Furthermore, it reaches a maze completion rate of 80.6% with PPO and 79.2% with A2C, again taking the lead on direct training (77.8%). Interpolation showed more promise with the input recognition metric, although the low overall variations of this metric preclude additional inferences.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Aggregated maximum and median performance by trainer and algorithm. The best values for a row are indicated in bold and second-best in italic. EDHuCAT produced the most general maze navigation agent with respect to normalized rewards and maze completion rates (top rows). Interpolation and EDHuCAT show complementary capabilities to produce better maze navigation <italic>in general</italic> (bottom rows).</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Trainer</bold></th>
<th valign="top" align="center" colspan="2"><bold>Direct</bold></th>
<th valign="top" align="center" colspan="2"><bold>Interpolation</bold></th>
<th valign="top" align="center" colspan="2"><bold>EDHuCAT</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Algorithm</bold></td>
<td valign="top" align="center"><bold>A2C</bold></td>
<td valign="top" align="center"><bold>PPO</bold></td>
<td valign="top" align="center"><bold>A2C</bold></td>
<td valign="top" align="center"><bold>PPO</bold></td>
<td valign="top" align="center"><bold>A2C</bold></td>
<td valign="top" align="center"><bold>PPO</bold></td>
</tr> <tr>
<td valign="top" align="left">Maximum</td>
<td valign="top" align="center">Normalized reward</td>
<td valign="top" align="center">&#x02013;2.58</td>
<td valign="top" align="center"><italic>&#x02212;0.873</italic></td>
<td valign="top" align="center">&#x02013;2.51</td>
<td valign="top" align="center">&#x02013;1.2</td>
<td valign="top" align="center"><bold>&#x02212;0.498</bold></td>
<td valign="top" align="center"><bold>&#x02212;0.498</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Success rate</td>
<td valign="top" align="center">0.347</td>
<td valign="top" align="center">0.778</td>
<td valign="top" align="center">0.403</td>
<td valign="top" align="center">0.708</td>
<td valign="top" align="center"><italic>0.792</italic></td>
<td valign="top" align="center"><bold>0.806</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Input recognition</td>
<td valign="top" align="center">0.753</td>
<td valign="top" align="center">0.777</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center"><bold>0.785</bold></td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center"><italic>0.781</italic></td>
</tr> <tr>
<td valign="top" align="left">Median</td>
<td valign="top" align="center">Normalized reward</td>
<td valign="top" align="center">&#x02013;2.8</td>
<td valign="top" align="center">&#x02013;3.39</td>
<td valign="top" align="center">&#x02013;3.02</td>
<td valign="top" align="center"><bold>&#x02212;1.57</bold></td>
<td valign="top" align="center">&#x02013;2.87</td>
<td valign="top" align="center"><italic>&#x02212;1.87</italic></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Success rate</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.326</td>
<td valign="top" align="center">0.257</td>
<td valign="top" align="center"><italic>0.618</italic></td>
<td valign="top" align="center">0.514</td>
<td valign="top" align="center"><bold>0.667</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="center">Input recognition</td>
<td valign="top" align="center">0.743</td>
<td valign="top" align="center">0.715</td>
<td valign="top" align="center">0.729</td>
<td valign="top" align="center"><bold>0.755</bold></td>
<td valign="top" align="center">0.717</td>
<td valign="top" align="center"><italic>0.747</italic></td>
</tr></tbody>
</table>
</table-wrap>
<p>Complementarily, in the context of easily generating general maze-navigating agents, the median performance is useful to highlight which combination of training regime and algorithm was better across replicates. Although slightly less favorable for EDHuCAT, which is in the top position once and second position twice, the results still speak volumes in favor of nonstationary environments. However, this time around, PPO is clearly identifiable as the algorithm that performs the best, since EDHuCAT also shows a marked bias in its favor.</p>
</sec>
<sec>
<title>5.4 Human impact</title>
<p>The previous metrics showed how agents resulting from the EDHuCAT algorithm can have a wide range of performance. To provide a tentative investigation of the reasons for this variability, we classified the decisions made by the human agent into three categories: <italic>Careful</italic>, challenges are slowly integrated once previous ones are solved; <italic>Risky</italic>, the agent is exposed to unfair conditions to promote resilience; <italic>Moderate</italic>, new challenges can be presented even if the agent has not solved the previous ones. The results (in the associated record) show that the <italic>Careful</italic> strategy provides better performance. Agents resulting from both the PPO algorithm and this strategy often end at the top, while agents training with A2C followed an inverse trend.</p></sec>
</sec>
<sec id="s6">
<title>6 Conclusion and discussion</title>
<p>In this work, we presented a benchmark generator that is geared toward the easy generation of feature-specific mazes and the intuitive understanding of the resulting agents&#x00027; strategies. The visual cues (either pre-processed or raw) these agents must learn to use to successfully navigate mazes are designed in a CPU-friendly manner so as to drastically limit computational time. By grounding an embodied visual task in what is essentially a succession of lookup-table queries, we allow complex cognitive processes to take place while avoiding the cost of a full robotics simulator. As the agents have only access to local information, this generator is applicable across a broad range of research domains, e.g., from sequential decision making to embodied AI. To help future researchers in manipulating and comparing mazes with widely different characteristics, we introduced two partially orthogonal metrics that accurately capture two key features of such mazes: their Surprisingness and Deceptiveness.</p>
<p>Furthermore, to demonstrate the potential of this generator, we compared the training capabilities of the Advantage Actor-Critic (A2C) and Proximal Policy Optimization (PPO) algorithms in three different training regimens with varying levels of environmental diversity. Direct training was a brute-force approach with only a target maze, while the Interpolation case relied on a scaffolding approach presenting increasing challenges. Finally, an interactive methodology (EDHuCAT) was introduced to leverage human expertise as often as possible.</p>
<p>We evaluated the performance of both the maze navigation capabilities of trained agents and their ability to correctly process the entire observation space. Across all these metrics, it was shown that PPO significantly outperforms A2C in dynamic environments, demonstrating the relevance of the former in producing generalized agents. Furthermore, we found that EDHuCAT together with PPO was clearly one step above the alternatives when aiming for <italic>a</italic> general maze-navigating agent. At the same time, if one strives for more than a singular champion but, instead, for reproducible performance, then results point to both the Interpolation and interactive training setups as valid contenders when used in conjunction with PPO.</p>
<p>While demonstrating the potential of AMaze as a benchmark generator for AI agents, this work also raised a number of questions. First, we aim to confirm whether the observed higher performance of PPO is explained by its use of a trust region, which reduces learning speed and, in turn, overfitting. Furthermore, as we limited the study to two RL algorithms and a single neural architecture, many questions remain open with respect to the best choice of hyperparameters or even the applicability of other techniques, such as Evolutionary Algorithms. Second, we only briefly mentioned the impact of the human in the interactive case, and while preliminary data (Godin-Dubois, <xref ref-type="bibr" rid="B10">2024</xref>) show tentative relationships between the human strategy, the training algorithm, and performance, dedicated studies are required to provide definitive answers. The strategy could be studied, as well as additional factors: Do youngsters train better than their elders? Does having a background in AI help? Or can laymen outperform experts?</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s7">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/10622914">https://zenodo.org/records/10622914</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>KG-D: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. KM: Conceptualization, Formal analysis, Funding acquisition, Methodology, Project administration, Supervision, Validation, Writing &#x02013; review &#x00026; editing. AK: Conceptualization, Investigation, Methodology, Supervision, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This research was funded by the Hybrid Intelligence Center, a 10-year programme funded by the Dutch Ministry of Education, Culture and Science through the Netherlands Organisation for Scientific Research, <ext-link ext-link-type="uri" xlink:href="https://hybrid-intelligence-centre.nl">https://hybrid-intelligence-centre.nl</ext-link>, grant number 024.004.022.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript.</p></sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec><sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2025.1511712/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frai.2025.1511712/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<fn-group>
<fn id="fn0001"><p><sup>1</sup>AMaze library is available on PyPI at <ext-link ext-link-type="uri" xlink:href="https://pypi.org/project/amaze-benchmarker/">https://pypi.org/project/amaze-benchmarker/</ext-link>. The code for the experiment described thereafter is hosted at <ext-link ext-link-type="uri" xlink:href="https://github.com/kgd-al/amaze_edhucat_2024">https://github.com/kgd-al/amaze_edhucat_2024</ext-link>.</p></fn>
<fn id="fn0002"><p><sup>2</sup>Environment-Driven Evolutionary Selection (Godin-Dubois et al., <xref ref-type="bibr" rid="B11">2020</xref>), used for automated open-ended evolution.</p></fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Alaguna</surname> <given-names>C.</given-names></name> <name><surname>Gomez</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Maze benchmark for testing evolutionary algorithms,&#x0201D;</article-title> in <source>Proceedings of the Genetic and Evolutionary Computation Conference Companion</source> (<publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>ACM</publisher-name>), <fpage>1321</fpage>&#x02013;<lpage>1328</lpage>. <pub-id pub-id-type="doi">10.1145/3205651.3208285</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barto</surname> <given-names>A. G.</given-names></name> <name><surname>Sutton</surname> <given-names>R. S.</given-names></name> <name><surname>Anderson</surname> <given-names>C. W.</given-names></name></person-group> (<year>1983</year>). <article-title>Neuronlike adaptive elements that can solve difficult learning control problems</article-title>. <source>IEEE Trans. Syst. Man Cybern</source>. <volume>13</volume>, <fpage>834</fpage>&#x02013;<lpage>846</lpage>. <pub-id pub-id-type="doi">10.1109/TSMC.1983.6313077</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Beattie</surname> <given-names>C.</given-names></name> <name><surname>K&#x000F6;ppe</surname> <given-names>T.</given-names></name> <name><surname>Du&#x000E9; nez-Guzm&#x000E1;n</surname> <given-names>E. A.</given-names></name> <name><surname>Leibo</surname> <given-names>J. Z.</given-names></name></person-group> (<year>2020</year>). <article-title>DeepMind Lab2D</article-title>. <source>arXiv [Preprint]</source>. arXiv:2011.07027.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Beattie</surname> <given-names>C.</given-names></name> <name><surname>Leibo</surname> <given-names>J. Z.</given-names></name> <name><surname>Teplyashin</surname> <given-names>D.</given-names></name> <name><surname>Ward</surname> <given-names>T.</given-names></name> <name><surname>Wainwright</surname> <given-names>M.</given-names></name> <name><surname>K&#x000FC;ttler</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>DeepMind Lab</article-title>. <source>arXiv [Preprint]</source>. arXiv:1612.03801.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bellemare</surname> <given-names>M. G.</given-names></name> <name><surname>Naddaf</surname> <given-names>Y.</given-names></name> <name><surname>Veness</surname> <given-names>J.</given-names></name> <name><surname>Bowling</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). <article-title>The arcade learning environment: an evaluation platform for general agents</article-title>. <source>J. Artif. Intell. Res</source>. <volume>47</volume>, <fpage>253</fpage>&#x02013;<lpage>279</lpage>. <pub-id pub-id-type="doi">10.1613/jair.3912</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benjamini</surname> <given-names>Y.</given-names></name> <name><surname>Hochberg</surname> <given-names>Y.</given-names></name></person-group> (<year>1995</year>). <article-title>Controlling the false discovery rate: a practical and powerful approach to multiple testing</article-title>. <source>J. Roy. Statist. Soc. Ser. B</source> <volume>57</volume>, <fpage>289</fpage>&#x02013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id></citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cobbe</surname> <given-names>K.</given-names></name> <name><surname>Hesse</surname> <given-names>C.</given-names></name> <name><surname>Hilton</surname> <given-names>J.</given-names></name> <name><surname>Schulman</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Leveraging procedural generation to benchmark reinforcement learning,&#x0201D;</article-title> in <source>37th International Conference on Machine Learning, ICML 2020</source>, <fpage>2026</fpage>&#x02013;<lpage>2034</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Eiben</surname> <given-names>A.</given-names></name> <name><surname>Smith</surname> <given-names>J.</given-names></name></person-group> (<year>2015</year>). <source>Introduction to Evolutionary Computing, volume 28</source>. <publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer Berlin Heidelberg</publisher-name>. <pub-id pub-id-type="doi">10.1007/978-3-662-44874-8</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fujimoto</surname> <given-names>S.</given-names></name> <name><surname>Van Hoof</surname> <given-names>H.</given-names></name> <name><surname>Meger</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Addressing function approximation error in actor-critic methods,&#x0201D;</article-title> in <source>35th International Conference on Machine Learning, ICML 2018</source>, <fpage>2587</fpage>&#x02013;<lpage>2601</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Godin-Dubois</surname> <given-names>K.</given-names></name></person-group> (<year>2024</year>). <article-title>AMaze: Fully discrete training with three regimes (direct, scaffolding, interactive) and two algorithms (A2C, PPO)</article-title>. <source>arXiv [Preprint]</source>. arXiv:2411.13072v1.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Godin-Dubois</surname> <given-names>K.</given-names></name> <name><surname>Cussat-Blanc</surname> <given-names>S.</given-names></name> <name><surname>Duthen</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Beneficial catastrophes: leveraging abiotic constraints through environment-driven evolutionary selection,&#x0201D;</article-title> in <source>2020 IEEE Symposium Series on Computational Intelligence (SSCI)</source>, 94&#x02013;101. <pub-id pub-id-type="doi">10.1109/SSCI47803.2020.9308411</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Harries</surname> <given-names>L.</given-names></name> <name><surname>Lee</surname> <given-names>S.</given-names></name> <name><surname>Rzepecki</surname> <given-names>J.</given-names></name> <name><surname>Hofmann</surname> <given-names>K.</given-names></name> <name><surname>Devlin</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;MazeExplorer: a customisable 3D benchmark for assessing generalisation in reinforcement learning,&#x0201D;</article-title> in <source>2019 IEEE Conference on Games (CoG)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>4</lpage>. <pub-id pub-id-type="doi">10.1109/CIG.2019.8848048</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Henderson</surname> <given-names>P.</given-names></name> <name><surname>Islam</surname> <given-names>R.</given-names></name> <name><surname>Bachman</surname> <given-names>P.</given-names></name> <name><surname>Pineau</surname> <given-names>J.</given-names></name> <name><surname>Precup</surname> <given-names>D.</given-names></name> <name><surname>Meger</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Deep reinforcement learning that matters,&#x0201D;</article-title> in <source>Proceedings of the AAAI Conference on Artificial Intelligence</source>, 3207&#x02013;3214. <pub-id pub-id-type="doi">10.1609/aaai.v32i1.11694</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Juliani</surname> <given-names>A.</given-names></name> <name><surname>Khalifa</surname> <given-names>A.</given-names></name> <name><surname>Berges</surname> <given-names>V.-P.</given-names></name> <name><surname>Harper</surname> <given-names>J.</given-names></name> <name><surname>Teng</surname> <given-names>E.</given-names></name> <name><surname>Henry</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>&#x0201C;Obstacle tower: a generalization challenge in vision, control, and planning,&#x0201D;</article-title> in <source>Proceedings of the Twenty-Eighth International Joint Conference on Artificial Intelligence, volume August</source> (<publisher-loc>California</publisher-loc>: <publisher-name>International Joint Conferences on Artificial Intelligence Organization</publisher-name>), <fpage>2684</fpage>&#x02013;<lpage>2691</lpage>. <pub-id pub-id-type="doi">10.24963/ijcai.2019/373</pub-id></citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kempka</surname> <given-names>M.</given-names></name> <name><surname>Wydmuch</surname> <given-names>M.</given-names></name> <name><surname>Runc</surname> <given-names>G.</given-names></name> <name><surname>Toczek</surname> <given-names>J.</given-names></name> <name><surname>Jaskowski</surname> <given-names>W.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;ViZDoom: a Doom-based AI research platform for visual reinforcement learning,&#x0201D;</article-title> in <source>2016 IEEE Conference on Computational Intelligence and Games (CIG)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>1</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1109/CIG.2016.7860433</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Laskin</surname> <given-names>M.</given-names></name> <name><surname>Yarats</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Zhan</surname> <given-names>A.</given-names></name> <name><surname>Lu</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;URLB: unsupervised reinforcement learning benchmark,&#x0201D;</article-title> in <source>NeurIPS</source>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mnih</surname> <given-names>V.</given-names></name> <name><surname>Badia</surname> <given-names>A. P.</given-names></name> <name><surname>Mirza</surname> <given-names>M.</given-names></name> <name><surname>Graves</surname> <given-names>A.</given-names></name> <name><surname>Lillicrap</surname> <given-names>T. P.</given-names></name> <name><surname>Harley</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Asynchronous methods for deep reinforcement learning</article-title>. <source>arXiv [Preprint]</source>. arXiv:1602.01783.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mnih</surname> <given-names>V.</given-names></name> <name><surname>Kavukcuoglu</surname> <given-names>K.</given-names></name> <name><surname>Silver</surname> <given-names>D.</given-names></name> <name><surname>Rusu</surname> <given-names>A. A.</given-names></name> <name><surname>Veness</surname> <given-names>J.</given-names></name> <name><surname>Bellemare</surname> <given-names>M. G.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Human-level control through deep reinforcement learning</article-title>. <source>Nature</source> <volume>518</volume>, <fpage>529</fpage>&#x02013;<lpage>533</lpage>. <pub-id pub-id-type="doi">10.1038/nature14236</pub-id><pub-id pub-id-type="pmid">25719670</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nichol</surname> <given-names>A.</given-names></name> <name><surname>Pfau</surname> <given-names>V.</given-names></name> <name><surname>Hesse</surname> <given-names>C.</given-names></name> <name><surname>Klimov</surname> <given-names>O.</given-names></name> <name><surname>Schulman</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>Gotta learn fast: a new benchmark for generalization in RL</article-title>. <source>arXiv [Preprint]</source>. arXiv:1804.03720.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Raffin</surname> <given-names>A.</given-names></name> <name><surname>Hill</surname> <given-names>A.</given-names></name> <name><surname>Gleave</surname> <given-names>A.</given-names></name> <name><surname>Kanervisto</surname> <given-names>A.</given-names></name> <name><surname>Ernestus</surname> <given-names>M.</given-names></name> <name><surname>Dormann</surname> <given-names>N.</given-names></name></person-group> (<year>2021</year>). <article-title>Stable-baselines3: reliable reinforcement learning implementations</article-title>. <source>J. Mach. Learn. Res</source>. <volume>22</volume>, <fpage>1</fpage>&#x02013;<lpage>8</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Risi</surname> <given-names>S.</given-names></name> <name><surname>Togelius</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Increasing generality in machine learning through procedural content generation</article-title>. <source>Nat. Mach. Intell</source>. <volume>2</volume>, <fpage>428</fpage>&#x02013;<lpage>436</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-020-0208-z</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schulman</surname> <given-names>J.</given-names></name> <name><surname>Wolski</surname> <given-names>F.</given-names></name> <name><surname>Dhariwal</surname> <given-names>P.</given-names></name> <name><surname>Radford</surname> <given-names>A.</given-names></name> <name><surname>Klimov</surname> <given-names>O.</given-names></name></person-group> (<year>2017</year>). <article-title>Proximal policy optimization algorithms</article-title>. <source>arXiv [Preprint]</source>. arXiv:1707.06347.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shannon</surname> <given-names>C. E.</given-names></name></person-group> (<year>1948</year>). <article-title>A mathematical theory of communication</article-title>. <source>Bell Syst. Techn. J</source>. <volume>27</volume>, <fpage>379</fpage>&#x02013;<lpage>423</lpage>. <pub-id pub-id-type="doi">10.1002/j.1538-7305.1948.tb01338.x</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sutton</surname> <given-names>R.</given-names></name> <name><surname>Barto</surname> <given-names>A.</given-names></name></person-group> (<year>2018</year>). <source>Reinforcement Learning: An Introduction</source>. <publisher-loc>Bradford</publisher-loc>: <publisher-name>A Bradford Book</publisher-name>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Synnaeve</surname> <given-names>G.</given-names></name> <name><surname>Nardelli</surname> <given-names>N.</given-names></name> <name><surname>Auvolat</surname> <given-names>A.</given-names></name> <name><surname>Chintala</surname> <given-names>S.</given-names></name> <name><surname>Lacroix</surname> <given-names>T.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>TorchCraft: a library for machine learning research on real-time strategy games</article-title>. <source>arXiv [Preprint]</source>. arXiv:1611.00625.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tian</surname> <given-names>Y.</given-names></name> <name><surname>Gong</surname> <given-names>Q.</given-names></name> <name><surname>Shang</surname> <given-names>W.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Zitnick</surname> <given-names>C. L.</given-names></name></person-group> (<year>2017</year>). <article-title>ELF: an extensive, lightweight and flexible research platform for real-time strategy games</article-title>. <source>arXiv [Preprint]</source>. arXiv:1707.01067.</citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Todorov</surname> <given-names>E.</given-names></name> <name><surname>Erez</surname> <given-names>T.</given-names></name> <name><surname>Tassa</surname> <given-names>Y.</given-names></name></person-group> (<year>2012</year>). <article-title>&#x0201C;MuJoCo: a physics engine for model-based control,&#x0201D;</article-title> in <source>2012 IEEE/RSJ International Conference on Intelligent Robots and Systems</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>5026</fpage>&#x02013;<lpage>5033</lpage>. <pub-id pub-id-type="doi">10.1109/IROS.2012.6386109</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tomilin</surname> <given-names>T.</given-names></name> <name><surname>Dai</surname> <given-names>T.</given-names></name> <name><surname>Fang</surname> <given-names>M.</given-names></name> <name><surname>Pechenizkiy</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;LevDoom: a benchmark for generalization on level difficulty in reinforcement learning,&#x0201D;</article-title> in <source>2022 IEEE Conference on Games (CoG)</source> (<publisher-loc>IEEE</publisher-loc>), <fpage>72</fpage>&#x02013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1109/CoG51982.2022.9893707</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Towers</surname> <given-names>M.</given-names></name> <name><surname>Kwiatkowski</surname> <given-names>A.</given-names></name> <name><surname>Terry</surname> <given-names>J.</given-names></name> <name><surname>Balis</surname> <given-names>J. U.</given-names></name> <name><surname>De Cola</surname> <given-names>G.</given-names></name> <name><surname>Deleu</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>Gymnasium: a standard interface for reinforcement learning environments</article-title>. <source>arXiv [Preprint]</source>. arXiv:2407.17032.</citation>
</ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>R.</given-names></name> <name><surname>Lehman</surname> <given-names>J.</given-names></name> <name><surname>Clune</surname> <given-names>J.</given-names></name> <name><surname>Stanley</surname> <given-names>K. O.</given-names></name></person-group> (<year>2019</year>). <article-title>Paired open-ended trailblazer (POET): endlessly generating increasingly complex and diverse learning environments and their solutions</article-title>. <source>arXiv [Preprint]</source>. arXiv:1901.01753.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wawrzy&#x00144;ski</surname> <given-names>P.</given-names></name></person-group> (<year>2009</year>). <article-title>&#x0201C;A cat-like robot real-time learning to run,&#x0201D;</article-title> in <source>Adaptive and Natural Computing Algorithms. ICANNGA 2009</source>, 380&#x02013;390. <pub-id pub-id-type="doi">10.1007/978-3-642-04921-7_39</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>T.</given-names></name> <name><surname>Quillen</surname> <given-names>D.</given-names></name> <name><surname>He</surname> <given-names>Z.</given-names></name> <name><surname>Julian</surname> <given-names>R.</given-names></name> <name><surname>Narayan</surname> <given-names>A.</given-names></name> <name><surname>Shively</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Meta-world: a benchmark and evaluation for multi-task and meta reinforcement learning</article-title>. <source>arXiv [Preprint]</source>. arXiv:1910.10897.</citation>
</ref>
</ref-list>
</back>
</article>