<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1221739</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2023.1221739</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Learning to reason over scene graphs: a case study of finetuning GPT-2 into a robot language model for grounded task planning</article-title>
<alt-title alt-title-type="left-running-head">Chalvatzaki et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2023.1221739">10.3389/frobt.2023.1221739</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Chalvatzaki</surname>
<given-names>Georgia</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/809325/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Younes</surname>
<given-names>Ali</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2339952/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Nandha</surname>
<given-names>Daljeet</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Le</surname>
<given-names>An Thai</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ribeiro</surname>
<given-names>Leonardo F. R.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="fn" rid="fn1">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2341469/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gurevych</surname>
<given-names>Iryna</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Computer Science Department</institution>, <institution>Technische Universit&#xe4;t Darmstadt</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Hessian.AI</institution>, <addr-line>Darmstadt</addr-line>, <country>Germany</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Center for Mind, Brain and Behavior</institution>, <institution>University Marburg and JLU Giessen</institution>, <addr-line>Marburg</addr-line>, <country>Germany</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Amazon Alexa</institution>, <addr-line>Seattle</addr-line>, <addr-line>WA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/386852/overview">Dimitrios Kanoulas</ext-link>, University College London, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/802205/overview">Maria Koskinopoulou</ext-link>, Heriot-Watt University, United Kingdom</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1275567/overview">Dario Albani</ext-link>, Technology Innovation Institute (TII), United Arab Emirates</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Georgia Chalvatzaki, <email>georgia.chalvatzaki@tu-darmstadt.de</email>
</corresp>
<fn fn-type="present-address" id="fn1">
<label>
<sup>&#x2020;</sup>
</label>
<p>
<bold>Present address:</bold> Leonardo F. R. Ribeiro, TU Darmstadt, Darmstadt, Germany</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>15</day>
<month>08</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>10</volume>
<elocation-id>1221739</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>05</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>03</day>
<month>07</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Chalvatzaki, Younes, Nandha, Le, Ribeiro and Gurevych.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Chalvatzaki, Younes, Nandha, Le, Ribeiro and Gurevych</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Long-horizon task planning is essential for the development of intelligent assistive and service robots. In this work, we investigate the applicability of a smaller class of large language models (LLMs), specifically GPT-2, in robotic task planning by learning to decompose tasks into subgoal specifications for a planner to execute sequentially. Our method grounds the input of the LLM on the domain that is represented as a scene graph, enabling it to translate human requests into executable robot plans, thereby learning to reason over long-horizon tasks, as encountered in the ALFRED benchmark. We compare our approach with classical planning and baseline methods to examine the applicability and generalizability of LLM-based planners. Our findings suggest that the knowledge stored in an LLM can be effectively grounded to perform long-horizon task planning, demonstrating the promising potential for the future application of neuro-symbolic planning methods in robotics.</p>
</abstract>
<kwd-group>
<kwd>robot learning</kwd>
<kwd>task planning</kwd>
<kwd>grounding</kwd>
<kwd>language models (LMs)</kwd>
<kwd>pretrained models</kwd>
<kwd>scene graphs</kwd>
</kwd-group>
<contract-num rid="cn001">CH2676/1-1</contract-num>
<contract-sponsor id="cn001">Deutsche Forschungsgemeinschaft<named-content content-type="fundref-id">10.13039/501100001659</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Field Robotics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>The autonomous execution of long-horizon tasks is of utmost importance for future assistive and service robots. An intelligent robot should reason about its surroundings, e.g., regarding the included objects and their spatial-semantic relations, and abstract an action plan for achieving a goal that will purposefully alter the perceived environment. Such an elaborate course of robot actions requires scene understanding, semantic reasoning, and planning over symbols and geometries. The advent of Deep Learning led many researchers to faithfully follow end-to-end approaches due to the representation power of differentiable deep neural networks (<xref ref-type="bibr" rid="B29">LeCun et al., 2015</xref>).</p>
<p>The problem of sequential decision-making has been addressed both with search-based and optimization approaches (<xref ref-type="bibr" rid="B25">Kaelbling and Lozano-P&#xe9;rez, 2011</xref>; <xref ref-type="bibr" rid="B46">Toussaint, 2015</xref>; <xref ref-type="bibr" rid="B10">Driess and Toussaint, 2019</xref>; <xref ref-type="bibr" rid="B15">Garrett et al., 2021</xref>; <xref ref-type="bibr" rid="B16">2020</xref>), as well as learning-based (<xref ref-type="bibr" rid="B34">Nair and Finn, 2019</xref>; <xref ref-type="bibr" rid="B13">Funk et al., 2021</xref>; <xref ref-type="bibr" rid="B18">Hoang et al., 2021</xref>) and hybrid methods (<xref ref-type="bibr" rid="B26">Kim et al., 2019</xref>; <xref ref-type="bibr" rid="B9">Driess et al., 2020</xref>; <xref ref-type="bibr" rid="B40">Ren et al., 2021</xref>; <xref ref-type="bibr" rid="B14">Funk et al., 2022</xref>). While the first ones enjoy probabilisticcompleteness, they require full domain specification and have high computational demands.The learning-based methods require broad exploration to learn from experience, but they have shown better generalization capabilities in similar domains to those experienced during training.</p>
<p>Large Language Models (LLMs) have exhibited an unprecedented generative ability (<xref ref-type="bibr" rid="B4">Bommasani et al., 2021</xref>), thanks to the transformer architecture (<xref ref-type="bibr" rid="B48">Vaswani et al., 2017</xref>) combined with massive datasets distilled from the internet. Naturally, in the quest for general artificial intelligence, researchers try to benchmark such models in reasoning tasks, among others (<xref ref-type="bibr" rid="B50">Wang et al., 2018</xref>; <xref ref-type="bibr" rid="B49">2019</xref>). Robotic embodied intelligence requires both logical and geometric reasoning; hence, it is a holy grail of AI. Several researchers saw a benefit in LLMs, and it was not long before several works explored their application to robotics for endowing robots with reasoning abilities in the scope of autonomous task planning and interaction (<xref ref-type="bibr" rid="B5">Brohan et al., 2022</xref>; <xref ref-type="bibr" rid="B52">Wei et al., 2022b</xref>). However, most works have focused on the prompting (<xref ref-type="bibr" rid="B6">Brown et al., 2020</xref>) and the subsequent prompt engineering (<xref ref-type="bibr" rid="B55">White et al., 2023</xref>), in which engineers provide appropriate inputs to LLMs for extracting outputs that can be realizable by a robotic agent, either for human-instruction following (<xref ref-type="bibr" rid="B35">Ouyang et al., 2022</xref>) or for planning (<xref ref-type="bibr" rid="B44">Singh et al., 2022</xref>; <xref ref-type="bibr" rid="B58">Zeng et al., 2022</xref>).</p>
<p>In this work, we study a finetuning process for grounding a small LLM for robotics, i.e., GPT-2 (<xref ref-type="bibr" rid="B38">Radford et al., 2021</xref>), to be used as a high-level abstraction in a task planning pipeline. Particularly, we propose a method that decomposes a long-horizon task into subgoals in the form of goal specifications for a robotic task planner to execute, and we investigate whether such a method can reach the performance levels of an oracle task planning baseline.</p>
<p>Our contribution is twofold: (i) we propose a novel method for linearizing the relations in a scene-graph structure representing the domain (world) to provide it as grounding context when finetuning a pretrained language model (e.g., GPT-2) for learning to draw associations between possible actions (goto, pick, etc.) and objects in the scene (e.g., kitchen, apple, etc.). Importantly, in our context, we encode the relative position of objects (far, close, right, left, etc.), allowing our model to account for the scene&#x2019;s geometrical structure when learning to plan. The proper structure of the input context is necessary for enabling the model to reason about the combinatorics of actions with affordable objects and their logical sequence (e.g., to cook something, one must first go to the kitchen). (ii) We showed that larger pretrained models do not necessarily possess grounded reasoning abilities, while it is possible to finetune smaller models on various tasks to use them as parts of a broader neuro-symbolic planning architecture. Contrarily to works that directly apply actions suggested by the GPTs to robots, we use language models at a higher level of abstraction, effectively suggesting sub-goals as PDDL problems to be solved by a Fast Downward task planner <xref ref-type="bibr" rid="B17">Helmert (2006)</xref>, effectively decomposing the whole problem into smaller ones of lower complexity.</p>
<p>Our thorough experimental evaluation shows that finetuning GPT-2 by additionally grounding its input on the domain can help translate human requests (tasks) to executable robot plans, to learn to reason over long-horizon tasks, as those encountered in the ALFRED benchmark (<xref ref-type="bibr" rid="B43">Shridhar et al., 2020</xref>). We compare our proposed approach with classical planning methods to investigate the applicability and generalizability of the Pre-trained Language Model (PLM)-based planners compared to classical task planners operating on a limited computational budget for a fair comparison. We conclude that the knowledge stored in a PLM can be grounded on different domains to perform long-horizon task planning, showing encouraging results for the future application of neuro-symbolic planning methods in robotics.</p>
</sec>
<sec id="s2">
<title>2 State of the art</title>
<sec id="s2-1">
<title>2.1 Reasoning with large language models</title>
<p>LLMs have attracted much attention for understanding the commonsense and reasoning patterns in their latent space (<xref ref-type="bibr" rid="B59">Zhou et al., 2020</xref>; <xref ref-type="bibr" rid="B31">Li et al., 2021</xref>; <xref ref-type="bibr" rid="B3">Bian et al., 2023</xref>). It has been shown that some abilities in logical and mathematical reasoning seem to emerge when LLMs are prompted appropriately (<xref ref-type="bibr" rid="B52">Wei et al., 2022b</xref>; <xref ref-type="bibr" rid="B51">a</xref>). However, the engineering effort, as well as the lack of robustness, is a key issue in prompting massive models (<xref ref-type="bibr" rid="B41">Ruis et al., 2022</xref>; <xref ref-type="bibr" rid="B47">Valmeekam et al., 2022</xref>). While great effort seems to be consumed on few-shot prompting of huge parametric models, it has also been shown by other lines of work show that efficient finetuning of much smaller models (<xref ref-type="bibr" rid="B45">Tay et al., 2022</xref>), or the use of small adaptation modules (Adapters) (<xref ref-type="bibr" rid="B20">Houlsby et al., 2019</xref>; <xref ref-type="bibr" rid="B37">Pfeiffer et al., 2021</xref>) can lead to methods that perform more robustly than large-scale generalist few-shot prompters. In the same direction, the chatbot versions of those huge models raised several points of criticism recently, showing that much more is needed than just prompting a blind human-preference alignment<xref ref-type="fn" rid="fn2">
<sup>1</sup>
</xref>.</p>
</sec>
<sec id="s2-2">
<title>2.2 Robot behavior planning</title>
<p>Long-horizon robot behavior planning is an NP-hard problem (<xref ref-type="bibr" rid="B53">Wells et al., 2019</xref>). Current advances in ML and perception led researchers to revisit this fundamental problem, i.e., the execution of multi-stage tasks, whose completion requires many sequential goals to be achieved, considering learning-based heuristics (<xref ref-type="bibr" rid="B9">Driess et al., 2020</xref>). Researchers consider such problems as Task And Motion Planning (TAMP) problems (<xref ref-type="bibr" rid="B15">Garrett et al., 2021</xref>; <xref ref-type="bibr" rid="B40">Ren et al., 2021</xref>; <xref ref-type="bibr" rid="B57">Xu et al., 2022</xref>), with a symbolic plan over entities and predicate with respective action operators with preconditions and effects in the environment. In contrast, a motion plan tries to find a feasible path to the goal. Nevertheless, most TAMP methods rely on manually specified rules; they do not integrate perception, and the combinatorial explosion when searching over symbolic and continuous parameters prohibits scaling the methods to challenging, realistic problems (<xref ref-type="bibr" rid="B26">Kim et al., 2019</xref>; <xref ref-type="bibr" rid="B16">Garrett et al., 2020</xref>).</p>
<p>Transformer models (<xref ref-type="bibr" rid="B48">Vaswani et al., 2017</xref>) that revolutionized the field of Natural Language Processing (NLP) opened the way for multiple new applications, in particular for robotics, e.g., visual-language instruction following (<xref ref-type="bibr" rid="B36">Pashevich et al., 2021</xref>), 3D scene understanding and grounding (<xref ref-type="bibr" rid="B8">Chen W. et al., 2022</xref>; <xref ref-type="bibr" rid="B33">Mees et al., 2022</xref>), language-based navigation (<xref ref-type="bibr" rid="B21">Huang C. et al., 2022</xref>; <xref ref-type="bibr" rid="B42">Shah et al., 2023</xref>). Due to their training on extensive databases, several works explored the use of LLMs for task planning and long-horizon manipulation (<xref ref-type="bibr" rid="B22">Huang et al., 2022b</xref>), mainly employing clever prompting (<xref ref-type="bibr" rid="B39">Raman et al., 2022</xref>; <xref ref-type="bibr" rid="B44">Singh et al., 2022</xref>), using multimodal information (<xref ref-type="bibr" rid="B24">Jiang et al., 2022</xref>; <xref ref-type="bibr" rid="B58">Zeng et al., 2022</xref>), grounding with value-functions (<xref ref-type="bibr" rid="B7">Chen B. et al., 2022</xref>; <xref ref-type="bibr" rid="B5">Brohan et al., 2022</xref>; <xref ref-type="bibr" rid="B23">Huang et al., 2022c</xref>), and deploying advances in code generation to extract executable robot plans (<xref ref-type="bibr" rid="B32">Liang et al., 2022</xref>). (<xref ref-type="bibr" rid="B30">Li et al., 2022</xref>) propose to use a PLM as a scaffold for decision-making policies in interactive environments, demonstrating benefits in the generalization abilities for policy learning even when language is not provided as input or output. Recently, PALM-e (<xref ref-type="bibr" rid="B9">Driess et al., 2020</xref>) has integrated a vision transformer with the PALM language model and has encoded some robotic state data to propose a multimodal embodied model, which showed the potential of integrating geometric information of the robot state but achieved limited performance in robotic tasks.</p>
</sec>
</sec>
<sec id="s3">
<title>3 Language models for grounded robot task planning</title>
<sec id="s3-1">
<title>3.1 Problem statement</title>
<p>Let us assume an agent that is able to move and manipulate objects in an environment, e.g., a mobile manipulator robot in a household environment. Let the environment be composed of a combination of <italic>rooms</italic>, such as &#x2018;bathroom,&#x2019; &#x2018;living room,&#x2019; or &#x2018;kitchen.&#x2019; Each room contains <italic>objects</italic> and <italic>receptacles</italic>, i.e., objects that are able to receive other objects, such as &#x2018;table,&#x2019; &#x2018;drawer,&#x2019; or &#x2018;sink.&#x2019; Each object has (household-specific) properties (affordances) associated with it that define whether it can be picked up, cleaned, heated, cooled, cut, etc. These properties can change, meaning that the objects have a <italic>state</italic>. The agent can pick up only one object at a time, meaning the agent also has a state, e.g., &#x2018;object in hand.&#x2019; Given the fact that the state is preserved over time and future actions depend on past actions, the environment can be characterized as sequential. Therefore, a series of actions has to be reasoned upon for an agent to be able to execute a series of actions for solving a long-horizon task, i.e., a task that requires the completion of several subtasks and potentially the manipulation of various objects to achieve the end-goal.</p>
</sec>
<sec id="s3-2">
<title>3.2 The ALFRED benchmark</title>
<p>The ALFRED benchmark (<xref ref-type="bibr" rid="B43">Shridhar et al., 2020</xref>) contains human-annotated training samples and image-based recordings of everyday household tasks; this is &#x201c;25,743 English language directives describing 8,055 expert demonstrations averaging 50 steps each, resulting in 428,322 image-action pair&#x201d;. In addition to that, the dataset provides a <italic>PDDL domain</italic> of the overall task and a <italic>PDDL problem</italic> for each sample (<xref ref-type="bibr" rid="B1">Aeronautiques et al., 1998</xref>). ALFRED heavily depends on AI2-THOR (<xref ref-type="bibr" rid="B28">Kolve et al., 2017</xref>), which acts as the underlying controller and simulation environment (based on the Unity game engine): trajectories for each sample of the ALFRED dataset were generated with AI2-THOR, and the validation of user-generated actions requires the AI2-THOR controller. <xref ref-type="fig" rid="F1">Figure 1</xref> shows a sample scene loaded into the AI2-THOR simulator. Each sample in the dataset consists of a high-level plan in PDDL and the trajectory of the agent&#x2019;s actions which lead to successful task completion, together with a description of the task goal and each plan step in Natural Language (NL).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>AI2-THOR simulator rendering a sample rollout from the ALFRED (<xref ref-type="bibr" rid="B43">Shridhar et al., 2020</xref>). The scenes show a room with household objects and the robot executing a task. Note that the robot does not have an arm, and the object automatically floats in front of the camera; it interacts with the environment through discrete actions. The discrete actions are shown underneath each frame in the form of PDDL commands.</p>
</caption>
<graphic xlink:href="frobt-10-1221739-g001.tif"/>
</fig>
</sec>
<sec id="s3-3">
<title>3.3 State and action space</title>
<p>The state space defines the feedback provided by the environment, while the action space defines the available actions to interact with the environment.</p>
<p>
<bold>State space.</bold> The environments we consider in this work are fully observable, having access to the complete simulator state representing the domain, as commonly considered for task planning tasks. AI2-THOR, the underlying simulator, eliminates most physics-related aspects (e.g., objects are automatically picked up and placed by a single action), which makes the highly dynamic and stochastic household environment almost static and deterministic&#x2014;almost because some physics still exists. This simplifies the core TAMP problem along with the discrete agent actions defined in the ALFRED dataset. Therefore, the ALFRED benchmark represents an appropriate choice for studying the problem of learning for robotic task planning, where motion failures are minimized by the underlying AI2-THOR controllers. Hence, we can focus on the reasoning aspects of the problem, which is the focus of this study. In the following, the state of the domain is transformed to NL context (&#xa7;3.6) and not used directly as model input.</p>
<p>
<bold>Action space.</bold> ALFRED has an action space of eight discrete high-level actions: <italic>GotoLocation</italic>, <italic>PickupObject</italic>, <italic>PutObject</italic>, <italic>CoolObject</italic>, <italic>HeatObject</italic>, <italic>CleanObject</italic>, <italic>SliceObject</italic> and <italic>ToggleObject</italic>. The underlying AI2-THOR navigation controller also has a discrete action space; the agent can <bold>move</bold> <italic>forward</italic>, <italic>backward</italic>, <italic>left</italic> or <italic>right</italic> and <bold>rotate</bold> <italic>clockwise</italic> or <italic>counter-clockwise</italic> in fixed steps.</p>
</sec>
<sec id="s3-4">
<title>3.4 Task categories</title>
<p>The ALFRED dataset encompasses seven categories of household tasks: &#x201c;Look at object,&#x201d; &#x201c;Pick and place,&#x201d; &#x201c;Pick two and place,&#x201d; &#x201c;Pick and place with movable receptacle,&#x201d; &#x201c;Pick, clean then place,&#x201d; &#x201c;Pick, cool then place,&#x201d; &#x201c;Pick, heat then place.&#x201d; Because objects can be placed in different corners of a room, each of these tasks includes the sub-problem of navigation. For the &#x2018;pick&#x2019; or &#x2018;place&#x2019; subtasks executing the respective <italic>PickupObject</italic> or <italic>PutObject</italic> action is sufficient. But, the subtasks &#x201c;clean&#x201d;, &#x201c;cool&#x201d; and &#x201c;heat&#x201d; must be seen as planning problems on their own, because the corresponding actions are a composition of high-level <italic>state-dependent</italic> actions. Regarding the household environment, the subtask &#x201c;cool&#x201d; requires a fridge, &#x201c;heat&#x201d; requires a microwave (or oven), and &#x201c;clean&#x201d; requires a sink as an <italic>receptacle</italic>. The ALFRED simulator tracks the state of each object, and the subtask is only considered successful when the final object state is correct. For example, if the task category is &#x2018;Pick, clean then place&#x2019;, the task goal is only completed when the placed object is marked as &#x2018;clean.&#x2019; The implementation aspects of these task categories are discussed in <xref ref-type="sec" rid="s4">Section 4</xref>.</p>
</sec>
<sec id="s3-5">
<title>3.5 RobLM: robot language model for task plan generation</title>
<p>Just like images can be represented by discretizing color space, NL can be expressed as a sequence of tokens <bold>x</bold> &#x3d; [<italic>x</italic>
<sub>1</sub>, <italic>x</italic>
<sub>2</sub>, &#x2026; , <italic>x</italic>
<sub>
<italic>n</italic>
</sub>], where each token is mapped to an embedding (lookup table). The Language Model (LM) places a probability distribution <italic>p</italic>(<bold>x</bold>) over the output token sequence. <italic>p</italic>(<bold>x</bold>) can be decomposed into a conditional probability distribution <italic>p</italic> (<italic>x</italic>
<sub>
<italic>i</italic>&#x2b;1</sub>&#x7c;<italic>x</italic>
<sub>
<italic>i</italic>
</sub>), where the probability of each token depends on all previous tokens. This results in the following joint distribution <italic>p</italic>(<bold>x</bold>) for a sequence of tokens <bold>x</bold>:<disp-formula id="e1">
<mml:math id="m1">
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mo>&#x220f;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
<p>In regards to Neural Network (NN), <italic>p</italic>(<bold>x</bold>) is commonly estimated with the Softmax function (<xref ref-type="bibr" rid="B2">Bengio et al., 2000</xref>)<xref ref-type="fn" rid="fn3">
<sup>2</sup>
</xref>
<disp-formula id="e2">
<mml:math id="m2">
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>exp</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">W</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">T</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="bold">b</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>where <bold>W</bold> is the learned weight matrix, <bold>b</bold> the bias and <bold>h</bold>
<sup>
<bold>T</bold>
</sup> the output vector of the NN. For text generation, the joint probability distribution <italic>p</italic>(<bold>x</bold>) (see Eq. <xref ref-type="disp-formula" rid="e1">1</xref>) can be formulated as a maximum-likelihood objective, where the objective is to maximize the likelihood of the next token occurrence for the given data.</p>
<p>Our goal is to finetune a LM to get a Robot Language Model (RobLM) that can generate a complete high-level task plan in one shot, given the domain information and a task goal. Because LMs are unsupervised learners, a single training sample contains both given and desired information as NL text. A restriction to the text format (a string of characters) comes with challenges: structural information needs to be condensed into a single linear dimension, and conceptually different aspects of the input need to be annotated in the text. This text format, including the syntax, has to be designed in such a way that information can be fed to and extracted from the LM reliably.</p>
<p>In RobLM, the format definition for a NL task description must comply with the following syntactic rule (spaces added for readability):</p>
<p>
<monospace>Goal [&#x3c;SEP&#x3e; Context] &#x3c;BOS&#x3e; Plan &#x3c;EOS&#x3e;</monospace>
</p>
<p>
<monospace>[...] :&#x3d; optional</monospace>
</p>
<p>
<monospace>&#x3c;SEP&#x3e; :&#x3d; separator token</monospace>
</p>
<p>
<monospace>&#x3c;BOS&#x3e; :&#x3d; begin-of-sequence token</monospace>
</p>
<p>
<monospace>&#x3c;EOS&#x3e; :&#x3d; end-of-sequence token</monospace>
</p>
<p>
<bold>Goal</bold> is the task goal in NL. <bold>Context</bold> is any additional, yet optional information provided to the LM. The task might have ambiguous solutions, and the inherent assumption is that the LM will better &#x201c;understand&#x201d; the task if given a context. Examples of a context are the name of the <italic>room</italic>, the name of the target object, or a NL description of the environment (see 3.6).</p>
<p>
<bold>Plan</bold> is the sequence of high-level task actions and their respective arguments, such as objective name or location. Because PLM have been trained on a diverse corpus of NL, including program code, the format for plans follows syntactical rules similar to that of a generic programming language:</p>
<p>
<monospace>Action0(arg0[,arg1]); Action1(arg0[,arg1]); ...</monospace>
</p>
<p>The sequence between the special tokens <inline-formula id="inf1">
<mml:math id="m3">
<mml:mo>&#x3c;</mml:mo>
<mml:mtext>BOS</mml:mtext>
<mml:mo>&#x3e;</mml:mo>
</mml:math>
</inline-formula> and <inline-formula id="inf2">
<mml:math id="m4">
<mml:mo>&#x3c;</mml:mo>
<mml:mtext>EOS</mml:mtext>
<mml:mo>&#x3e;</mml:mo>
</mml:math>
</inline-formula> can be extracted to retrieve the plan from the LM-generated output.</p>
<sec id="s3-5-1">
<title>3.5.1 Data augmentation</title>
<p>Each sample in the ALFRED dataset can be replayed in the AI2-THOR simulator to collect additional information not contained in the original dataset. ALFRED provides a script that has been modified for that purpose. Data augmentation is necessary for Graph2NL (c.f. &#xa7;3.6) to generate a graph representation from the environment state. For each replayed sample, the complete list of objects in the scene, with their respective name, position, and rotation, and the agent position is saved to a separate file next to the trajectory data. This file is later loaded and turned into a processable graph.</p>
</sec>
</sec>
<sec id="s3-6">
<title>3.6 Mapping scene graphs to natural language: Graph2NL</title>
<p>PLMs are trained on NL. Because of this, NL is a natural modality for finetuning a PLM. When a context is provided to the LM, this context must be presented in NL just like the input sequence. If the context should encapsulate the environment state, this means that the state has to be transformed into NL before being supplied to the PLM.</p>
<p>Graph2NL is a novel method that &#x201c;translates&#x201d; the object-centric scene graph representation of the environment state to NL. Optionally, domain knowledge about the environment<xref ref-type="fn" rid="fn4">
<sup>3</sup>
</xref> can be infused into this graph. The following steps describe the core Graph2NL process.<list list-type="simple">
<list-item>
<p>1) Generate an object-scene graph <italic>G</italic> with a node for the agent and nodes corresponding to objects, node attributes being the position and rotation of the object in Euclidean space and their respective distance and orientation vectors as edge attributes.</p>
</list-item>
<list-item>
<p>2) (Optional) Infuse domain knowledge about the environment by connecting all dependent nodes and all nodes reachable by the agent.</p>
</list-item>
<list-item>
<p>3) Connect the agent (node) to all reachable nodes, if given domain knowledge, or to all nodes, if not given domain knowledge.</p>
</list-item>
<list-item>
<p>4) Given a task and the identified target object, find all paths in the graphs leading from the agent (node) to the target object (node).</p>
</list-item>
<list-item>
<p>5) Use edge attributes in the found paths to describe the task-centric environment state, by mapping geometric relations to NL tokens.</p>
</list-item>
</list>
</p>
<sec id="s3-6-1">
<title>3.6.1 NL mapping</title>
<p>To translate geometric relations attributed by the graph edges into a NL description, a mapping function is designed. In human speech, distances are expressed by a vocabulary of words such as &#x201c;close&#x201d; or &#x201c;far&#x201d; and orientations are expressed by words such as &#x201c;in front&#x201d; or &#x201c;behind&#x201d;. Graph2NL adapts this vocabulary to describe the (numeric) distance and orientation from one node relative to another in NL.</p>
<p>
<xref ref-type="table" rid="T1">Table 1</xref> summarizes the mapping used in Graph2NL. The distance between nodes is expressed in Cartesian space and orientation in polar coordinates, where <italic>Yaw</italic> is the azimuth angle (rotation along the surface normal) and <italic>Pitch</italic> is the zenith angle (altitude). With this mapping, the geometric relation between two nodes can be explained by three words (one for each: distance, pitch, and yaw). The vocabulary contains 8 words to express the distance, 4 words to express the vertical, and 2 words to express the horizontal orientation. Combinatorially, this gives 64 possible geometric configurations. The geometric relationship is expressed in a condensed form by treating each of these configurations as a relation and assigning a special symbol (token) for each relation. A simple approach, referring to the <italic>Symbol</italic> column in <xref ref-type="table" rid="T1">Table 1</xref>, is by assigning a symbol to each word. Combining the symbols for distance, pitch, and yaw creates the condensed (three-letter) representation of the geometric relationship. These symbolic representations can optionally be added to the LM tokenizer as <italic>special</italic> tokens. Shorter token sequences generally decrease both the training and inference time.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Graph2NL mapping table. Distances are mapped to NL vocabulary (or a symbol) in a one-to-one relation. <italic>Yaw</italic> describes the orientation along the surface normal when viewed from a top-down perspective, and <italic>Pitch</italic> describes the z-planar offset (altitude) in relation to the origin.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">Distance [m]</th>
</tr>
<tr>
<th align="center">Value</th>
<th align="center">NL</th>
<th align="center">Symbol</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">
<inline-formula id="inf3">
<mml:math id="m5">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>5</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">distant</td>
<td align="center">a</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf4">
<mml:math id="m6">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>4</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">far</td>
<td align="center">b</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf5">
<mml:math id="m7">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>3</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">reachable</td>
<td align="center">c</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf6">
<mml:math id="m8">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>2</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">near</td>
<td align="center">d</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf7">
<mml:math id="m9">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">close</td>
<td align="center">e</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf8">
<mml:math id="m10">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">closer</td>
<td align="center">f</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf9">
<mml:math id="m11">
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">next</td>
<td align="center">g</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf10">
<mml:math id="m12">
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0.1</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">in</td>
<td align="center">h</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">Yaw [&#xb0;]</th>
</tr>
<tr>
<th align="center">Value</th>
<th align="center">NL</th>
<th align="center">Symbol</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">45 to 135</td>
<td align="center">right</td>
<td align="center">i</td>
</tr>
<tr>
<td align="center">135 to 225</td>
<td align="center">back</td>
<td align="center">j</td>
</tr>
<tr>
<td align="center">225 to 315</td>
<td align="center">left</td>
<td align="center">k</td>
</tr>
<tr>
<td align="center">315 to 45</td>
<td align="center">front</td>
<td align="center">l</td>
</tr>
</tbody>
</table>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">Pitch [&#xb0;]</th>
</tr>
<tr>
<th align="center">Value</th>
<th align="center">NL</th>
<th align="center">Symbol</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">
<inline-formula id="inf11">
<mml:math id="m13">
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">above</td>
<td align="center">m</td>
</tr>
<tr>
<td align="center">
<inline-formula id="inf12">
<mml:math id="m14">
<mml:mo>&#x3c;</mml:mo>
<mml:mn>0</mml:mn>
</mml:math>
</inline-formula>
</td>
<td align="center">below</td>
<td align="center">n</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<bold>Example</bold> Let the task be: &#x201c;Put the soap into the drawer&#x201d;. The input query to Graph2NL consists of the target object &#x201c;soap&#x201d;. <xref ref-type="fig" rid="F2">Figure 2</xref> shows the graph constructed by Graph2NL from augmented data (&#xa7;3.5.1), including domain-specific knowledge.After finding the shortest paths between the root (&#x2018;agent&#x2019;) and target node (&#x2018;soapbar&#x2019;), Graph2NL produces an output in the following form (cut-off at search depth 2):</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Graph2NL example graph. After locating the root (&#x201c;agent&#x201d;) and target node (&#x201c;soapbar&#x201d;), the shortest paths connecting those nodes are found and summarized in NL by mapping all edge attributes along the path.</p>
</caption>
<graphic xlink:href="frobt-10-1221739-g002.tif"/>
</fig>
<p>
<monospace>[Bathroom&#x3d;</monospace>
</p>
<p>
<monospace>- closer below left sink near below back soapbar</monospace>
</p>
<p>
<monospace>- closer below left cabinet near above back soapbar</monospace>
</p>
<p>
<monospace>- closer above left countertop next above back soapbar</monospace>
</p>
<p>
<monospace>- close below back toilet closer below back soapbar</monospace>
</p>
<p>
<monospace>- closer below back garbagecan close below back soapbar]</monospace>
</p>
<p>The NL context by Graph2NL starts with the name of the room extracted from the scene graph, followed by the geometric description of each node connected to the target node on the path from the agent. &#x201c;-&#x201d; indicates the root note, i.e., the agent. For the previous example, Graph2NL produces the following <bold>condensed</bold> form:</p>
<p>
<monospace>[Bathroom&#x3d;</monospace>
</p>
<p>
<monospace>- fnk sink dnj soapbar</monospace>
</p>
<p>
<monospace>- fnj cabinet dmj soapbar</monospace>
</p>
<p>
<monospace>- fmk countertop gmj soapbar</monospace>
</p>
<p>
<monospace>- enj toilet fnj soapbar</monospace>
</p>
<p>
<monospace>- fnj garbagecan enj soapbar]</monospace>
</p>
<p>This form of state representation is unique for each problem configuration and forms the context that grounds RobLM.</p>
</sec>
</sec>
<sec id="s3-7">
<title>3.7 Training</title>
<p>RobLM generates a plan as text given the goal and the context, which involves causal language modeling for text generation. Decoder-only autoregressive language models (<xref ref-type="fig" rid="F3">Figure 3</xref>) are frequently used for the problem of text generation; we chose GPT-2 as the base model for RobLM. RobLM uses the base version of the GPT-2 PLM (&#x2018;gpt-2&#x2019;) (<xref ref-type="bibr" rid="B38">Radford et al., 2021</xref>), loaded and initialized with pre-trained weights from the Huggingface (<xref ref-type="bibr" rid="B56">Wolf et al., 2019</xref>) Transformer library. Finetuning GPT-2 for causal language generation has a self-supervised setup, where the labels are the inputs shifted to the right, which entitles learning to predict the next token in a sequence.We finetune the GPT-2 model using the pre-processed training data of the ALFRED dataset, which has around 20.000 samples, with three sets of NL descriptions for each sample. The ADAM (<xref ref-type="bibr" rid="B27">Kingma and Ba, 2014</xref>) optimizer is used with a learning rate of 5<italic>e</italic>
<sup>&#x2212;5</sup>, and the LM is trained for two epochs. Finetuning a GPT-2 LM to the ALFRED training data with a single GPU-accelerated computer takes around 30 min (27 iterations/s - measurement not representative due to hardware dependence).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Decoder-only Transformer architecture. The input to the decoder is tokenized text, and the output is probabilities over the tokens in the tokenizer vocabulary. The positional encoding is added to the embedded input to account for the order. Transformer&#x2019;s decoder can have multiple transformer blocks, each of which contains multi-head attention with linear layers and layer normalization.</p>
</caption>
<graphic xlink:href="frobt-10-1221739-g003.tif"/>
</fig>
</sec>
<sec id="s3-8">
<title>3.8 Generation pipeline</title>
<p>For inference, RobLM takes only the NL task goal together with an optional context and outputs the complete step-by-step plan for completing the goal. This plan is composed of high-level instructions rather than low-level controller commands.</p>
<p>
<bold>Example.</bold> Given the task &#x201c;Put the soap into the drawer:&#x201d;, RobLM (no context) generates the plan:</p>
<p>
<monospace>Put the soap into the drawer:</monospace>
</p>
<p>
<monospace>0.GotoLocation(countertop)</monospace>
</p>
<p>
<monospace>1.PickupObject(soap)</monospace>
</p>
<p>
<monospace>2.GotoLocation(drawer)</monospace>
</p>
<p>
<monospace>3.PutObject(soap,drawer)</monospace>
</p>
<p>The plan is generated by consecutive forward passes through the Transformer model. For a vocabulary size of <italic>k</italic> and a token sequence of length <italic>l</italic> (with <italic>l</italic> &#x2264; 1,024 for GPT-2), the forward pass of the Transformer yields an output vector of size <italic>k</italic> &#xd7; <italic>l</italic> with values in the interval [0, 1]. The Transformer outputs <italic>scores</italic>, i.e., <italic>logits</italic>, for each token in the input sequence. These scores are converted to a probability distribution <italic>p</italic>(<bold>x</bold>) by using the Softmax function, as described in Eq. <xref ref-type="disp-formula" rid="e2">2</xref>.</p>
<p>Two possible generation strategies for selecting the next token from <italic>p</italic>(<bold>x</bold>) are: <italic>greedy search</italic> and <italic>top-k/top-p sampling</italic> (<xref ref-type="bibr" rid="B19">Holtzman et al., 2020</xref>). In the greedy strategy, the token <italic>x</italic>
<sub>
<italic>sel</italic>
</sub> with the highest likelihood is picked with <italic>x</italic>
<sub>
<italic>sel</italic>
</sub> &#x3d; arg&#x2009;max&#x2009;<italic>p</italic>(<bold>x</bold>). In the top-k sampling strategy, as the name suggests, the scores are sorted, and one of the first <italic>k</italic> candidate tokens is randomly sampled. By extending the top-k sampling with an additional top-p strategy, the sum of the <italic>k</italic> candidates must be equal to or greater than <italic>p</italic> &#x2208; [0, 1]. Simply put, top-k widens the choice over the next tokens and top-p filters out low-probability tokens. <xref ref-type="fig" rid="F4">Figure 4</xref> illustrates, on the basis of an example, a forward pass through the Transformer with a greedy selection strategy. These steps are repeated recursively until an end-of-text token is encountered or the defined sequence limit is reached to generate the full plan.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Illustration of a forward pass through RobLM for text generation with a greedy next-token selection strategy. The forward passes are repeated in a recursive manner until an end-of-text token is encountered or the defined sequence limit is reached.</p>
</caption>
<graphic xlink:href="frobt-10-1221739-g004.tif"/>
</fig>
<p>This LM model was finetuned to generate a structured output, omitting special tokens, characterized by numbered actions and their arguments in parenthesis. The input is always part of the output, due to the generation function utilized by RobLM. Note that it is not guaranteed that the &#x2018;soap&#x2019; can be found inside the &#x2018;drawer&#x2019; on the &#x2018;countertop&#x2019;. In fact, it could be at any possible location permitted by the environment. However, given a greedy search strategy, for the given task goal, the <italic>likelihood</italic> for the &#x2018;soap&#x2019; being on the &#x2018;countertop&#x2019; is the highest in this case.</p>
<sec id="s3-8-1">
<title>3.8.1 Hardware setup</title>
<p>For finetuning LMs and evaluating each model, we used the Lichtenberg Cluster of TU Darmstadt, which contains stacks of NVIDIA<sup>&#xae;</sup> A100 and V100 GPUs. Internal tests have shown that a single GPU can decrease the training time by a factor of 10 (these tests are not representative because performance depends on every hardware component). To run experiments in the AI2-THOR simulation, we used a PC with an NVIDIA<sup>&#xae;</sup> RTX 3080Ti GPU.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments</title>
<sec id="s4-1">
<title>4.1 Preliminary analysis for task plan generation with GPT-2 and GPT-3</title>
<p>LLMs can represent knowledge from the data they have been trained on. However, the question remains, whether this knowledge can be leveraged to solve planning tasks, i.e., can LLMs reason? This is investigated by comparing the text-generation results of GPT-2 (Zero-Shot Learning (ZSL)) and GPT-3 (Few-Shot Learning (FSL)) for a planning task.</p>
<p>
<monospace>Given an instruction to a household robot,</monospace>
</p>
<p>
<monospace>formulate the steps to complete the instruction.</monospace>
</p>
<p>
<monospace>The instruction is: &#x2018;&#x2018;Put a washed slice of apple on the table.&#x2019;&#x2019;</monospace>
</p>
<p>The results for this task vary greatly between the two LLMs.</p>
<p>
<bold>GPT-2</bold> (&#x2018;gpt-2&#x2019; model, 1.5B parameters) completely fails this task and produces an output that resembles a repetition of phrases in the input sentence:</p>
<p>
<monospace>Put a washed slice of apple on the table.</monospace>
</p>
<p>
<monospace>Put a washed slice of apple on the table.</monospace>
</p>
<p>
<monospace>Put a washed slice of apple on the table.</monospace>
</p>
<p>
<monospace>...</monospace>
</p>
<p>Similar behavior has been observed for other LMs falling into the ZSL category. The input sentence is not an open-ended question and requires reasoning.</p>
<p>
<bold>GPT-3</bold> (&#x2018;text-davinci-002&#x2019; model, 175B parameters), when given the same instruction as input, is able to make assumptions and formulate a whole plan based on these assumptions:</p>
<p>
<monospace>Assuming the apple is not cut and not washed.</monospace>
</p>
<p>
<monospace>1. Pick up the apple from the counter.</monospace>
</p>
<p>
<monospace>2. Cut the apple into a slice.</monospace>
</p>
<p>
<monospace>3. Wash the apple slice.</monospace>
</p>
<p>
<monospace>4. Place the apple slice on the table.</monospace>
</p>
<p>The FSL paradigm allows GPT-3 to be very sensitive to context changes and seemingly <italic>understand</italic> the request at hand. However, smaller GPT-3 PLMs (GPT-3 curie, GPT-3 babbage, GPT3-ada) show a degraded quality in the produced plan (<xref ref-type="bibr" rid="B12">Floridi and Chiriatti, 2020</xref>). have shown that GPT-3 would not pass the Turing test, as to having &#x201c;no understanding of the semantics and contexts of the request, but only a syntactic (statistical) capacity to associate words [&#x2026;]&#x201d;.</p>
<p>These tests have shown that plan generation capabilities of LLMs vary dramatically depending on the underlying learning paradigm, model architecture, and parameter size. GPT-2, out of the box, is completely unsuited for solving planning tasks that require a minimum level of text understanding. However, as later (&#xa7;3.5) shown, GPT-2 can successfully generate plans when finetuned to a training dataset (&#xa7;3.2). The question of whether a finetuned GPT-2 model can <italic>leverage</italic> knowledge for planning is addressed in the following section. GPT-3, unfortunately, is only accessible through a paid service by OpenAI, and finetuning of own GPT-3 models is possible through the provided service. Practical applications, however, are limited because each query has to be sent to and processed by the OpenAI service. Even if a PLM was made available, the hardware requirements for running GPT-3 models are immense, even for today&#x2019;s standards, due to the sheer parameter count. It is for these reasons that GPT-3 and its newest versions are not considered as a basis for finetuning to RobLM.</p>
</sec>
<sec id="s4-2">
<title>4.2 Evaluation of RobLM</title>
<p>This section presents the main experiments conducted for evaluation of RobLM. We first define the appropriate metrics and a baseline method required to make the evaluations measurable and comparable. The <italic>grounding</italic> problem is explained in accordance with the practical aspects of integrating the available methods into the simulator. For the experimentation part, a set of finetuned LLMss is compared with the baseline performance.</p>
<sec id="s4-2-1">
<title>4.2.1 Metrics</title>
<p>To validate a finetuned LM, only the NL task goal of each validation sample and optionally, the context is fed to the RobLM generation pipeline (see <xref ref-type="fig" rid="F4">Figure 4</xref>). Validation is performed over each task category rather than all the validation data. This enables the analysis of a task-dependent performance: some task categories are more complex than others leading to a longer trajectory of actions and hence an increased difficulty. Two metrics are defined for validation: LM accuracy and plan success rate.</p>
<p>
<bold>Definition&#x2014;Accuracy</bold>. Accuracy measures how accurately the LM is able to predict the following parts of the plan.<list list-type="simple">
<list-item>
<p>&#x2022; the correct count and names of all actions in the plan (action accuracy)</p>
</list-item>
<list-item>
<p>&#x2022; the correct count and names of all arguments in the plan (argument accuracy)</p>
</list-item>
<list-item>
<p>&#x2022; the correct count and names of all actions <bold>and</bold> arguments in the plan (&#x201c;full plan&#x201d; accuracy)</p>
</list-item>
</list>For a found plan, the accuracy of actions and arguments counts if <bold>all</bold> actions or arguments are correct. With this metric, it is possible to anchor the cause of plan failure to either the actions or the arguments, or both.</p>
<p>Having an accurate LM does not necessarily mean that the generated plan leads to success&#x2014;at least, as long the &#x201c;full plan&#x201d; accuracy is below 1.0, i.e., the trajectory is not replicated perfectly. A second metric is required that measures the actual <italic>success rate</italic> of the finetuned LM in simulation. There are two possible scenarios that justify this additional metric. First, the plan could fail in simulation, even if it seems accurate. And second, the plan could succeed in simulation, even if the plan is not completely accurate.</p>
<p>
<bold>Definition&#x2014;Success rate</bold>. The success rate is a measure of the successful completion of individual sub-tasks of a validation task. After loading the trajectory, environment state, and goal from the validation sample into the AI2-THOR simulator, the actions predicted by the LM are translated into low-level controller actions via task and geometric grounding (&#xa7;4.2.3), which are then passed to the AI2-THOR controller and executed in the simulator. After every simulator step, a check is performed to determine whether the target conditions for sub-task completion have been met. If the target conditions are kept unsatisfied after execution of the last low-level action, it counts as a success towards the sub-task, or otherwise, as a failure.</p>
</sec>
<sec id="s4-2-2">
<title>4.2.2 Baseline</title>
<p>A baseline is an oracle, or upper bound, that serves as a measurement reference. Fast Downward (FD) (<xref ref-type="bibr" rid="B17">Helmert, 2006</xref>) is used as the baseline for evaluation. We consider a classical task planner like FD appropriate since it also has access to the full domain and is a complete algorithm (<xref ref-type="bibr" rid="B17">Helmert, 2006</xref>). Therefore, the ability of a RobLM to match or outperform FD (for a given time budget) would reveal whether LMs can be helpful towards learning task planning. Every ALFRED validation sample comes with a PDDL problem file, while the PDDL domain is shared by all tasks; this allows the PDDL planner to generate a plan for each sample. To generate a plan using FD, the PDDL problem files provided by ALFRED have to be pre-processed. FD is able to handle Action Description Language (ADL) instructions, as found in the PDDL problem, but is not able to process optimization-related additional information present in the files.</p>
</sec>
<sec id="s4-2-3">
<title>4.2.3 Instruction grounding</title>
<p>
<italic>Grounding</italic> can be defined as mapping a high-level, abstract, or symbolic representation to a low-level grounded representation. Grounding of an abstract plan to objects is called object or geometric grounding (or &#x201c;world grounding&#x201d;), and grounding of NL to robot tasks is called task grounding. In this case, instructions generated by the LM are made up of actions that require a <italic>task grounding</italic>, and arguments, which require a <italic>geometric grounding</italic>.</p>
</sec>
<sec id="s4-2-4">
<title>4.2.4 Task grounding</title>
<p>Plans generated by RobLM consist of high-level actions and are not directly executable by the AI2-THOR controller. Each possible action predicted by the LM has to be grounded to a task, which then translates to a sequence of low-level controller actions. For task grounding, three possible types of tasks are defined: navigation, manipulation and composite. In a navigation task, the agent is required to move from one to another location. In a manipulation task, the agent performs an action affecting the environment state. Composite tasks are a composition of manipulation tasks that need to be completed in a specific order.</p>
<p>Task grounding is performed as follows.<list list-type="simple">
<list-item>
<p>&#x2022; The action <italic>GotoLocation</italic> is grounded to the navigation task and delegated to a trajectory planner for navigation (see below).</p>
</list-item>
<list-item>
<p>&#x2022; The actions <italic>PickupObject</italic>, <italic>PutObject</italic>, <italic>ToggleObject</italic> and <italic>SliceObject</italic> are grounded to the manipulation task, the actions can be directly executed by the low-level controller.</p>
</list-item>
<list-item>
<p>&#x2022; The actions <italic>HeatObject</italic>, <italic>CoolObject</italic> and <italic>CleanObject</italic> are grounded to the composite task, which is translated to this sequence of low-level actions: <italic>ToggleObject</italic> &#x2192;<italic>PutObject</italic> &#x2192;<italic>ToggleObject</italic> &#x2192;<italic>ToggleObject</italic> &#x2192;<italic>PickupObject</italic> &#x2192;<italic>ToggleObject</italic> (example given below).</p>
</list-item>
</list>
</p>
</sec>
<sec id="s4-2-5">
<title>4.2.5 Geometric grounding</title>
<p>An argument can be either a location or an object name. An argument produced by the LM might be ambiguous or non-existing in the environment. In order to be understood by the controller, these arguments have to be grounded on a geometric level. For grounding arguments, first, all available objects are retrieved from the simulation. Then, the world coordinates of all objects matching the predicted symbol (target object) are gathered. E.g., if the predicted target object is &#x2018;soap&#x2019;, the position of all &#x2018;soap&#x2019;-type objects can be queried and retrieved from the simulator. The low-level control commands are finally generated with the help of the ground-truth navigation graph of the scene.</p>
</sec>
<sec id="s4-2-6">
<title>4.2.6 Navigation</title>
<p>By overlaying the world with a grid, every position in the world is given a discrete coordinate. A navigation graph (not to be confused with a <italic>scene graph</italic> or Graph2NL graph) creates a node for each coordinate and connects all the nodes that are <italic>accessible</italic> one from another. Similar to the procedure of Graph2NL (&#xa7;3.6), the navigation graph is traversed after locating the agent and target node by the object name. A search algorithm is used to find the shortest path in the graph from the agent to the target object - in this case, it is the A&#x2a; algorithm (<xref ref-type="bibr" rid="B11">Ducho&#x148; et al., 2014</xref>). The search returns a sequence of nodes, which corresponds to a sequence of coordinates (a trajectory). Lastly, a motion planner takes the trajectory as an input and outputs a sequence of low-level controller actions (AI2-THOR conveniently provides a motion planner for navigation).</p>
</sec>
<sec id="s4-2-7">
<title>4.2.7 Experimental results</title>
<p>A set of finetuned <bold>RobLM</bold> models are evaluated against the baseline. The finetuned models differ in the amount of context provided during training time.<list list-type="simple">
<list-item>
<p>1) &#x2018;No context&#x2019; &#x2014; Only task goal</p>
</list-item>
<list-item>
<p>2) &#x2018;Scene knowledge&#x2019; &#x2014; List of all available objects in the environment, found in the PDDL problem</p>
</list-item>
<list-item>
<p>3) &#x2018;Scene graph&#x2019; &#x2014; Description of geometric relations to the target object, generated by Graph2NL</p>
</list-item>
<list-item>
<p>4) &#x2018;Full context&#x2019; &#x2014; Description of geometric relations of all objects, generated by Graph2NL</p>
</list-item>
</list>Given a PDDL problem file, Graph2NL automatically generates the context in the specified text format. This context is provided to the LM for training and inference.</p>
<p>
<bold>Accuracy.</bold> <xref ref-type="fig" rid="F5">Figure 5</xref> summarizes the evaluation of the finetuned RobLM models compared to the FD baseline for previously unseen (validation) data. It can be observed that none of the finetuned RobLM models is able to outperform the baseline. Going through each of the models and starting with the &#x2018;No context&#x2019; model it is surprising that this model, even without any contextual information, is able to generate the correct plan actions with high accuracy. The &#x2018;Scene knowledge&#x2019; and &#x2018;Scene graph&#x2019; models have a similar performance, the &#x2018;Scene graph&#x2019; generally being slightly more accurate in both actions and argument prediction. Both of these models overall outperform the &#x2018;No context&#x2019; model, with a significant improvement of the context models in the arguments prediction.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Prediction accuracy of actions and arguments for previously <bold>unseen</bold> data across a set of tasks. Neither RobLM model is able to outperform the baseline (blue) but shows high accuracy in the prediction of plan actions. Context-driven models (green, red, and purple) perform better than the model without any scene-related context (orange).</p>
</caption>
<graphic xlink:href="frobt-10-1221739-g005.tif"/>
</fig>
<p>Given these results, the following conclusions about the examined models can be made.<list list-type="simple">
<list-item>
<p>1) Failed plans are mostly caused by wrong arguments (objects or locations) and only in some cases by wrong actions.</p>
</list-item>
<list-item>
<p>2) The LM is able to learn the structure of tasks, but not scene-dependent components.</p>
</list-item>
</list>
</p>
<p>RobLM is able to distinguish between the task categories and provide a correct task action plan. However, where this model fails is in finding all correct action arguments, i.e., locations and object names. This can be explained by the fact that the task goal alone does not reveal the actual location of the target object. Because the target can be in any accessible location in the environment, or in any accessible receptacle, the produced argument is the result of the LM imitating the <italic>most-likely</italic> cases observed in the training data.</p>
<p>Overall, these results are consistent with the point made on contextual information and prediction accuracy: giving to the model information about the environment, i.e., finetuning a model to be grounded to the scene, <bold>does</bold> improve performance.</p>
<p>
<bold>Success rate.</bold> A plan is successful if each of the sub-tasks for a stated task is completed. <xref ref-type="table" rid="T2">Table 2</xref> summarizes the success rate of RobLM compared to the baseline for actions of the navigation task (<italic>GotoLocation</italic>) and manipulation task (<italic>PickupObject</italic>, <italic>PutObject</italic>, etc.). Composite tasks have been omitted from this evaluation because of high failure rates caused by their task grounding complexity.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Success rates of sub-task completion in simulation&#x2014;RobLM (&#x2018;No context&#x2019;) compared to the baseline on seen and unseen validation data.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">
<italic>Success rate</italic>
</th>
<th colspan="2" align="center">Baseline</th>
<th colspan="2" align="center">RobLM (&#x2018;no context&#x2019;)</th>
</tr>
<tr>
<th align="center">Task</th>
<th align="center">seen</th>
<th align="center">unseen</th>
<th align="center">seen</th>
<th align="center">unseen</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">GotoLocation</td>
<td align="center">0.318</td>
<td align="center">0.393</td>
<td align="center">
<bold>0.422</bold>
</td>
<td align="center">
<bold>0.499</bold>
</td>
</tr>
<tr>
<td align="center">PickupObject</td>
<td align="center">0.466</td>
<td align="center">0.474</td>
<td align="center">
<bold>0.776</bold>
</td>
<td align="center">
<bold>0.749</bold>
</td>
</tr>
<tr>
<td align="center">PutObject</td>
<td align="center">
<bold>0.385</bold>
</td>
<td align="center">
<bold>0.331</bold>
</td>
<td align="center">0.116</td>
<td align="center">0.092</td>
</tr>
<tr>
<td align="center">SliceObject</td>
<td align="center">0.629</td>
<td align="center">0.5</td>
<td align="center">
<bold>0.94</bold>
</td>
<td align="center">
<bold>0.98</bold>
</td>
</tr>
<tr>
<td align="center">ToggleObject</td>
<td align="center">0</td>
<td align="center">0</td>
<td align="center">
<bold>0.84</bold>
</td>
<td align="center">
<bold>0.864</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Bold represent maximum values.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Regarding geometric grounding, arguments predicted by RobLM are grounded to all matching objects in the world, and RobLM is allowed to &#x201c;try&#x201d; all possibilities. E.g., when the objective is to &#x201c;Get soap&#x201d;, multiple &#x2018;soap&#x2019;-type objects could exist in the scene. Each possibility given by the geometric grounding is simulated by storing and restoring the simulator state. While this is a clear advantage for RobLM over the baseline, the evaluation still holds because the LM is required to predict the correct location or object names. Based on the presented results, the LM-based system performs well on sub-tasks requiring the action <italic>PickupObject</italic>, while the action <italic>PutObject</italic> does not succeed equally well, being far from the baseline performance.</p>
<p>Overall, the success rate of the baseline method is not nearly as high as expected, hinting at potential implementation-specific failures in the task grounding and in the low-level controller interaction with objects. In the low-level controller, visual information is not included. This means that the robot is controlled in a &#x201c;blind flight&#x201d; mode. The AI2-THOR simulation requires the target object to be in <italic>view</italic>. If the object is not visible, e.g., because the agent is looking in the wrong direction, the interaction fails and with it, the sub-task. Because of the fact that both systems have been evaluated within the same framework, these results <bold>do not</bold> dismiss a potential use-case for LM in planning.</p>
</sec>
<sec id="s4-2-8">
<title>4.2.8 Additional results</title>
<p>We provide additional experiments for a deeper analysis of potential points of failure of RobLM. These experiments entail a different sampling strategy and context refinement.</p>
<sec id="s4-2-8-1">
<title>4.2.8.1 Top-k/top-p sampling</title>
<p>So far, every experiment conducted has used a greedy next-token selection strategy. In order to be able to tell with certainty that a found plan is the &#x201c;best possible&#x201d; plan, a comparison with another sampling strategy is required. This additional experiment repeats the previous one, but this time with a top-k and top-p sampling strategy. The comparison is done with the &#x2018;No context&#x2019; RobLM model for all tasks with <italic>k</italic> &#x3d; 10 and <italic>p</italic> &#x3d; 0.9, i.e., tokens are sampled from the top-10 predictions and sum up to a probability <inline-formula id="inf13">
<mml:math id="m15">
<mml:mo>&#x2265;</mml:mo>
<mml:mn>0.9</mml:mn>
</mml:math>
</inline-formula>. Since a similar pattern was observed in the individual task evaluations, <xref ref-type="table" rid="T3">Table 3</xref> reports the results for the &#x2018;Pick Simple&#x2019; task only. Each token is sampled three times, giving three possible solutions to be evaluated. Slight&#x2014;uniform and hence dismissable&#x2014;variations in the prediction accuracy exist between these three runs. The sampling-based method performs slightly worse than the greedy strategy.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Top-k and top-p sampling (<italic>k</italic> &#x3d;10 and <italic>p</italic> &#x3d; 0.9) &#x2014; tokens are sampled three times for the &#x2018;Pick Simple&#x2019; task, giving only slight deviations in the final accuracy.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="4" align="center">RobLM &#x2018;no context&#x2019; model, &#x2018;pick simple&#x2019; task</th>
</tr>
<tr>
<th align="center">Accuracy of</th>
<th align="center">1st sample</th>
<th align="center">2nd sample</th>
<th align="center">3rd sample</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Actions</td>
<td align="center">0.7746</td>
<td align="center">0.8169</td>
<td align="center">0.7817</td>
</tr>
<tr>
<td align="center">Arguments</td>
<td align="center">0.3803</td>
<td align="center">0.4085</td>
<td align="center">0.3944</td>
</tr>
<tr>
<td align="center">GotoLocation</td>
<td align="center">0.8772</td>
<td align="center">0.9123</td>
<td align="center">0.8904</td>
</tr>
<tr>
<td align="center">PickupObject</td>
<td align="center">0.8380</td>
<td align="center">0.8662</td>
<td align="center">0.8451</td>
</tr>
<tr>
<td align="center">PutObject</td>
<td align="center">0.7971</td>
<td align="center">0.8227</td>
<td align="center">0.7986</td>
</tr>
<tr>
<td align="center">GotoLocation&#x5f;Args</td>
<td align="center">0.5658</td>
<td align="center">0.5877</td>
<td align="center">0.5833</td>
</tr>
<tr>
<td align="center">PickupObject&#x5f;Args</td>
<td align="center">0.7535</td>
<td align="center">0.8028</td>
<td align="center">0.7746</td>
</tr>
<tr>
<td align="center">PutObject&#x5f;Args</td>
<td align="center">0.6449</td>
<td align="center">0.6667</td>
<td align="center">0.6331</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4-2-8-2">
<title>4.2.8.2 Refined context</title>
<p>In a deeper analysis of RobLM failure cases, it has been found that the first argument in the generated plan is the hardest to predict correctly by the LM. The LM is not able to draw enough conclusions about the first instruction from the supplied context of any form. This causality becomes obvious after the following experiment: Given the task goal and a NL <bold>description</bold> of the first instruction as context, how does the overall accuracy of the LM change? The following text is an example of an instruction description in NL, as found in the ALFRED dataset:</p>
<p>
<monospace>Turn left and walk across the room towards the shelves on the wall.</monospace>
</p>
<p>The results in <xref ref-type="fig" rid="F6">Figure 6</xref> show that, given this extra information, RobLM is almost able to reach the performance levels of the baseline measurement across all tasks; it shows very high accuracy on &#x201c;full plan&#x201d; actions and arguments. The conclusion of this experiment is that the more precisely the supplied context is tailored towards the key issue of LM generation task, the more accurate the generated plan becomes. For this specific problem, finding the correct first argument is key to a successful plan, and with a NL description of the first instruction, the LM is able to draw the necessary connections from context to plan.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Prediction accuracy of actions and arguments for unseen tasks of RobLM with a refined context. This experiment compares a model finetuned to the task goal and a context consisting of a NL description of the first plan instruction (green) with the baseline (blue) and a RobLM with &#x2018;No context&#x2019; model (orange).</p>
</caption>
<graphic xlink:href="frobt-10-1221739-g006.tif"/>
</fig>
<p>The overall conclusion of this observation is that the LM are adaptive; the LM is able to adapt new information into the plan generation, towards a more accurate sequence of instructions.</p>
</sec>
</sec>
</sec>
<sec id="s4-3">
<title>4.3 Run-time analysis</title>
<p>Inference frequency is an important factor when it comes to real-life applications. This is especially true for industrial robotics, where cycle times are important. But not every robotic application is time-critical, e.g., a household robot is not expected to respond in a sub-second time. However, if task planning is seen as a programming problem, a fast execution time greatly enhances the operator experience <xref ref-type="table" rid="T4">Table 4</xref> shows a comparison of the inference speeds of RobLM against the baseline (FD). RobLM, in all cases, is slower compared to the baseline, which is likely due to the reliance on the full GPT-2 vocabulary size for the LM tokenizer and the usage of a LM-internal, implementation-specific generation function<xref ref-type="fn" rid="fn5">
<sup>4</sup>
</xref>. Such an issue can be mitigated by training a new tokenizer on the task-specific vocabulary, but this comes at the cost of not utilizing the stored knowledge in the PLM. However, current progress in language models allows faster inferences in more advanced hardware than the one used in this work; therefore, we believe that the frequency limitations can be easily overcome.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Comparison of inference speeds&#x2014;RobLM against baseline. GPU acceleration used for LM (NVIDIA<sup>&#xae;</sup> GeForce RTX 2080 SUPER). The timer starts only after the program or model has been loaded into memory, i.e., only computation (inference) time is measured. &#x201c;No context&#x201d; has a maximum token sequence length of 200 and &#x201c;Full context&#x201d; has a maximum length of 1,024 tokens for generation.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th colspan="3" align="center">Iterations per second (average over 800 samples)</th>
</tr>
<tr>
<th align="center">Baseline</th>
<th align="center">RobLM &#x2018;No context&#x2019;</th>
<th align="center">RobLM &#x2018;Full context&#x2019;</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">2.9</td>
<td align="center">1.0</td>
<td align="center">0.2</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>
<bold>Remarks.</bold> Overall, our analysis has shown that finetuning PLMs toward robotic task planning is possible when providing an appropriate grounding context. However, we have shown that such models cannot yet reach the planning abilities of classical task planners. A combination of finetuning with proper scene representation and a more elaborate sampling strategy, as well as the addition of more sophisticated prompts, can boost the performance of RobLM, leading them to performances that are closer to the oracle task planners. Still, the benefit of providing goal specifications as natural language commands alleviate the burden of engineering, while advances in scene graph generation can make the extraction of domain specifications autonomous. Therefore, we believe that using RobLMs at a higher level of abstraction for neuro-symbolic task planning is valuable but is still in its infancy. Additional challenges have been recently summarized by <xref ref-type="bibr" rid="B54">Weng (2023)</xref>, where some of the listed points are in accordance with our findings.</p>
</sec>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>We presented a framework for finetuning grounded Large Language Models (LLMs) and investigated the applicability of such models combined with planning in solving ling-horizon robot reasoning tasks. This paper has shown that LLMs can extract commonsense knowledge through precise queries and adjust their behavior based on available information or context. Among our contributions are the development of RobLM, a grounded finetuned LLM that generates plans directly from natural language commands, and Graph2NL, which creates natural language text describing graph-based data, to represent scene graphs as inputs into RobLM. Our extensive experimental results have revealed, nevertheless, the challenges in representing structured and geometric data in natural language. However, LLMs still need to demonstrate a consistent ability to perform long-horizon planning tasks and cannot yet replace classical planners. Despite their limitations, LLMs possess powerful features such as efficient storage and retrieval of commonsense knowledge, which can be useful in planning tasks when presented with partially observable environments.</p>
<p>For future work, exploring larger models like GPT-3 or GPT-NeoX could increase the accuracy and success rate of RobLM. Providing structured context to the Transformer model and exploring multi-modal inputs, such as visual information, may also improve the planning capabilities of LLMs. Further research in the field of applied natural language processing in robotics could help unlock the full potential of LLMs and contribute to the development of more advanced neuro-symbolic planning systems.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found below: <ext-link ext-link-type="uri" xlink:href="https://github.com/dnandha/RobLM">https://github.com/dnandha/RobLM</ext-link>.</p>
</sec>
<sec id="s7">
<title>Author contributions</title>
<p>GC, AY, and DN contributed to the conception and design of the method. AL assisted in the setup of the baseline method. LR provided insights for the training and fine-tuning of language models and advice on the linearization of the graph. IG gave advice on the overall method and the idea of using language models for planning. GC wrote the current version of the manuscript, building on the initial write-up by DN. AY assisted in writing and visualizations. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s8">
<title>Funding</title>
<p>This research has been supported by the German Research Foundation (DFG) through the Emmy Noether Programme (CH 2676/1-1) and by the Hessian. AI Connectom Fund &#x201c;Robot Learning of Long-Horizon Manipulation bridging Object-centric Representations to Knowledge Graphs&#x201d;. We acknowledge support by the Deutsche Forschungsgemeinschaft (DFG, German Research Foundation) and the Open Access Publishing Fund of Technical University of Darmstadt.</p>
</sec>
<ack>
<p>The authors would like to acknowledge Snehal Jauhri and Haau-Sing Li for the fruitful discussions and suggestions.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>Author LR was employed by Amazon Alexa.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn2">
<label>1</label>
<p>7 problems facing Bing, Bard, and the future of AI search, from The Verge.</p>
</fn>
<fn id="fn3">
<label>2</label>
<p>The Softmax function applies the exponential function to each element of the input vector and normalizes the values, by dividing by the sum of the exponentials.</p>
</fn>
<fn id="fn4">
<label>3</label>
<p>Domain knowledge entails every possible room, object, and receptacle name and their allowed relations, as described in the respective documentation: <ext-link ext-link-type="uri" xlink:href="https://ai2thor.allenai.org/ithor/documentation/objects/object-types">https://ai2thor.allenai.org/ithor/documentation/objects/object-types</ext-link>.</p>
</fn>
<fn id="fn5">
<label>4</label>
<p>Huggingface generation function: &#x2018;transformers/src/transformers/generation&#x5f;utils.py&#x2019;.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Aeronautiques</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Howe</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Knoblock</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>McDermott</surname>
<given-names>I. D.</given-names>
</name>
<name>
<surname>Ram</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Veloso</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>1998</year>). <article-title>Pddl&#x2013; the planning domain definition language</article-title>. <source>Tech. Rep. Tech. Rep.</source>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ducharme</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Vincent</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>A neural probabilistic language model</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>13</volume>.</citation>
</ref>
<ref id="B3">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Bian</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Chatgpt is a knowledgeable but inexperienced solver: an investigation of commonsense problem in large language models</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2303.16421">https://arxiv.org/abs/2303.16421</ext-link>
</comment>.</citation>
</ref>
<ref id="B4">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Bommasani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hudson</surname>
<given-names>D. A.</given-names>
</name>
<name>
<surname>Adeli</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Arora</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>von Arx</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>On the opportunities and risks of foundation models</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2108.07258">https://arxiv.org/abs/2108.07258</ext-link>
</comment>.</citation>
</ref>
<ref id="B5">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Brohan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chebotar</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hausman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Herzog</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Do as i can, not as i say: grounding language in robotic affordances</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2204.01691">https://arxiv.org/abs/2204.01691</ext-link>
</comment>.</citation>
</ref>
<ref id="B6">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Brown</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Mann</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ryder</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Subbiah</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kaplan</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Language models are few-shot learners</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2005.14165">https://arxiv.org/abs/2005.14165</ext-link>
</comment>.</citation>
</ref>
<ref id="B7">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ichter</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Rao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Gopalakrishnan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ryoo</surname>
<given-names>M. S.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Open-vocabulary queryable scene representations for real world planning</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2209.09874">https://arxiv.org/abs/2209.09874</ext-link>
</comment>.</citation>
</ref>
<ref id="B8">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Talak</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Carlone</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Leveraging large language models for robot 3d scene understanding</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2209.05629">https://arxiv.org/abs/2209.05629</ext-link>
</comment>.</citation>
</ref>
<ref id="B9">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Driess</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Ha</surname>
<given-names>J.-S.</given-names>
</name>
<name>
<surname>Toussaint</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Deep visual reasoning: learning to predict action sequences for task and motion planning from an initial scene image</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2006.05398">https://arxiv.org/abs/2006.05398</ext-link>
</comment>.</citation>
</ref>
<ref id="B10">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Driess</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Toussaint</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Hierarchical task and motion planning using logic-geometric programming (hlgp)</source>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ducho&#x148;</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Babinec</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kajan</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Be&#x148;o</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Florek</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fico</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Path planning with modified a star algorithm for a mobile robot</article-title>. <source>Procedia Eng.</source> <volume>96</volume>, <fpage>59</fpage>&#x2013;<lpage>69</lpage>. <pub-id pub-id-type="doi">10.1016/j.proeng.2014.12.098</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Floridi</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chiriatti</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Gpt-3: its nature, scope, limits, and consequences</article-title>. <source>Minds Mach.</source> <volume>30</volume>, <fpage>681</fpage>&#x2013;<lpage>694</lpage>. <pub-id pub-id-type="doi">10.1007/s11023-020-09548-1</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Funk</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Chalvatzaki</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Belousov</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Learn2assemble with structured representations and search for robotic architectural construction</article-title>. <source>Conf. Robot Learn. (CoRL)</source>.</citation>
</ref>
<ref id="B14">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Funk</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Menzenbach</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chalvatzaki</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Graph-based reinforcement learning meets mixed integer programs: an application to 3d robot assembly discovery</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2203.04120">https://arxiv.org/abs/2203.04120</ext-link>
</comment>.</citation>
</ref>
<ref id="B15">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Garrett</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Chitnis</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Holladay</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Integrated task and motion planning</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2010.01083">https://arxiv.org/abs/2010.01083</ext-link>
</comment>.</citation>
</ref>
<ref id="B16">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Garrett</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Lozano-P&#xe9;rez</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Pddlstream: integrating symbolic planners and blackbox samplers via optimistic adaptive planning</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1802.08705">https://arxiv.org/abs/1802.08705</ext-link>
</comment>.</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Helmert</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>The fast downward planning system</article-title>. <source>J. Artif. Intell. Res.</source> <volume>26</volume>, <fpage>191</fpage>&#x2013;<lpage>246</lpage>. <pub-id pub-id-type="doi">10.1613/jair.1705</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hoang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sohn</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Carvalho</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Successor feature landmarks for long-horizon goal-conditioned reinforcement learning</article-title>. <source>Adv. Neural Inf. Process. Syst.</source>
<volume>34</volume>, <fpage>26963</fpage>&#x2013;<lpage>26975</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Holtzman</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Buys</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Forbes</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>The curious case of neural text degeneration</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1904.09751">https://arxiv.org/abs/1904.09751</ext-link>
</comment>.</citation>
</ref>
<ref id="B20">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Houlsby</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Giurgiu</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Jastrzebski</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Morrone</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>De Laroussilhe</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Gesmundo</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Parameter-efficient transfer learning for nlp</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1902.00751">https://arxiv.org/abs/1902.00751</ext-link>
</comment>.</citation>
</ref>
<ref id="B21">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mees</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022a</year>). <article-title>Visual language maps for robot navigation</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2210.05714">https://arxiv.org/abs/2210.05714</ext-link>
</comment>.</citation>
</ref>
<ref id="B22">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Pathak</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Mordatch</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Language models as zero-shot planners: extracting actionable knowledge for embodied agents</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2201.07207">https://arxiv.org/abs/2201.07207</ext-link>
</comment>.</citation>
</ref>
<ref id="B23">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Florence</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022c</year>). <article-title>Inner monologue: embodied reasoning through planning with language models</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2207.05608">https://arxiv.org/abs/2207.05608</ext-link>
</comment>.</citation>
</ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Vima: general robot manipulation with multimodal prompts</article-title>. In <conf-name>Proceedings of the NeurIPS Foundation Models for Decision Making Workshop</conf-name>, <conf-loc>New Orleans, USA</conf-loc>, <conf-date>December 2022</conf-date>
</citation>
</ref>
<ref id="B25">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
<name>
<surname>Lozano-P&#xe9;rez</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Hierarchical task and motion planning in the now</article-title>. In <conf-name>Proceedings of the 2011 IEEE International Conference on Robotics and Automation</conf-name>, <fpage>1470</fpage>&#x2013;<lpage>1477</lpage>. <conf-loc>Shanghai, China</conf-loc>, <conf-date>May 2011</conf-date>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
<name>
<surname>Lozano-P&#xe9;rez</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Learning to guide task and motion planning using score-space representation</article-title>. <source>IJRR</source>
<volume>38</volume>, <fpage>793</fpage>&#x2013;<lpage>812</lpage>. <pub-id pub-id-type="doi">10.1177/0278364919848837</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Adam: A method for stochastic optimization</article-title>. <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1412.6980">https://arxiv.org/abs/1412.6980</ext-link>.</citation>
</ref>
<ref id="B28">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Kolve</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Mottaghi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>VanderBilt</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Weihs</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Herrasti</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Ai2-thor: an interactive 3d environment for visual ai</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1712.05474">https://arxiv.org/abs/1712.05474</ext-link>
</comment>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>Nature</source>
<volume>521</volume>, <fpage>436</fpage>&#x2013;<lpage>444</lpage>. <pub-id pub-id-type="doi">10.1038/nature14539</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Puig</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Paxton</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Fan</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Pre-trained language models for interactive decision-making</article-title>. <source>Adv. Neural Inf. Process. Syst.</source>
<volume>35</volume>, <fpage>31199</fpage>&#x2013;<lpage>31212</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X. L.</given-names>
</name>
<name>
<surname>Kuncoro</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>d&#x2019;Autume</surname>
<given-names>C. d. M.</given-names>
</name>
<name>
<surname>Blunsom</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Nematzadeh</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Do language models learn commonsense knowledge?</article-title> <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2111.00607">https://arxiv.org/abs/2111.00607</ext-link>
</comment>.</citation>
</ref>
<ref id="B32">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Hausman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ichter</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Code as policies: language model programs for embodied control</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2209.07753">https://arxiv.org/abs/2209.07753</ext-link>
</comment>.</citation>
</ref>
<ref id="B33">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Mees</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Borja-Diaz</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Grounding language with visual affordances over unstructured data</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2210.01911">https://arxiv.org/abs/2210.01911</ext-link>
</comment>.</citation>
</ref>
<ref id="B34">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Nair</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Hierarchical foresight: self-supervised learning of long-horizon tasks via visual subgoal generation</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1909.05829">https://arxiv.org/abs/1909.05829</ext-link>
</comment>.</citation>
</ref>
<ref id="B35">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Ouyang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Almeida</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Wainwright</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Mishkin</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Training language models to follow instructions with human feedback</article-title>. <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2203.02155">https://arxiv.org/abs/2203.02155</ext-link>.</citation>
</ref>
<ref id="B36">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Pashevich</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schmid</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Episodic transformer for vision-and-language navigation</article-title>. In <conf-name>Proceedings of the IEEE/CVF International Conference on Computer Vision</conf-name>. <conf-loc>Montreal, BC, Canada</conf-loc>, <conf-date>October 2021</conf-date>, <fpage>15942</fpage>&#x2013;<lpage>15952</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Pfeiffer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kamath</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>R&#xfc;ckl&#xe9;</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Cho</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Gurevych</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Adapterfusion: non-destructive task composition for transfer learning</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2005.00247">https://arxiv.org/abs/2005.00247</ext-link>
</comment>.</citation>
</ref>
<ref id="B38">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Hallacy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ramesh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Goh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Agarwal</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>&#x201c;Learning transferable visual models from natural language supervision</article-title>,&#x201d; in <conf-name>International conference on machine learning</conf-name> (<publisher-name>PMLR</publisher-name>), <fpage>8748</fpage>&#x2013;<lpage>8763</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Raman</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Cohen</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Rosen</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Idrees</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Paulius</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tellex</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Planning with large language models via corrective re-prompting</article-title>. <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2211.09935">https://arxiv.org/abs/2211.09935</ext-link>.</citation>
</ref>
<ref id="B40">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chalvatzaki</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Extended task and motion planning of long-horizon robot manipulation</article-title>. <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2103.05456">https://arxiv.org/abs/2103.05456</ext-link>.</citation>
</ref>
<ref id="B41">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Ruis</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Khan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Biderman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hooker</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rockt&#xe4;schel</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Grefenstette</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Large language models are not zero-shot communicators</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2210.14986">https://arxiv.org/abs/2210.14986</ext-link>
</comment>.</citation>
</ref>
<ref id="B42">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Shah</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Osi&#x144;ski</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Lm-nav: robotic navigation with large pre-trained models of language, vision, and action</article-title>,&#x201d; in <conf-name>Conference on robot learning</conf-name> (<publisher-name>PMLR</publisher-name>), <fpage>492</fpage>&#x2013;<lpage>504</lpage>.</citation>
</ref>
<ref id="B43">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Shridhar</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Thomason</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gordon</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bisk</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Han</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Mottaghi</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Alfred: A benchmark for interpreting grounded instructions for everyday tasks</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1912.01734">https://arxiv.org/abs/1912.01734</ext-link>
</comment>.</citation>
</ref>
<ref id="B44">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Blukis</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Mousavian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Tremblay</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Progprompt: generating situated robot task plans using large language models</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2209.11302">https://arxiv.org/abs/2209.11302</ext-link>
</comment>.</citation>
</ref>
<ref id="B45">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Tay</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chung</surname>
<given-names>H. W.</given-names>
</name>
<name>
<surname>Tran</surname>
<given-names>V. Q.</given-names>
</name>
<name>
<surname>So</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Shakeri</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Transcending scaling laws with 0.1% extra compute</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2210.11399">https://arxiv.org/abs/2210.11399</ext-link>
</comment>.</citation>
</ref>
<ref id="B46">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Toussaint</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Logic-geometric programming: an optimization-based approach to combined task and motion planning</article-title>.<conf-name>Proceedings of the Twenty-Fourth International Joint Conference on Artificial Intelligence (IJCAI 2015)</conf-name>, <conf-loc>Buenos Aires, Argentina</conf-loc>, <conf-date>July 2015</conf-date>.</citation>
</ref>
<ref id="B47">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Valmeekam</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Olmo</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sreedharan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kambhampati</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Large language models still can&#x2019;t plan (a benchmark for llms on planning and reasoning about change)</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2206.10498">https://arxiv.org/abs/2206.10498</ext-link>
</comment>.</citation>
</ref>
<ref id="B48">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname>
<given-names>A. N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Attention is all you need</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1706.03762">https://arxiv.org/abs/1706.03762</ext-link>
</comment>.</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Pruksachatkun</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Nangia</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michael</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hill</surname>
<given-names>F.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Superglue: A stickier benchmark for general-purpose language understanding systems</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>32</volume>.</citation>
</ref>
<ref id="B50">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Michael</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hill</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Levy</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Bowman</surname>
<given-names>S. R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Glue: A multi-task benchmark and analysis platform for natural language understanding</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1804.07461">https://arxiv.org/abs/1804.07461</ext-link>
</comment>.</citation>
</ref>
<ref id="B51">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tay</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bommasani</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Raffel</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zoph</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Borgeaud</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022a</year>). <article-title>Emergent abilities of large language models</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2206.07682">https://arxiv.org/abs/2206.07682</ext-link>
</comment>.</citation>
</ref>
<ref id="B52">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Wei</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Schuurmans</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bosma</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xia</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chi</surname>
<given-names>E. H.</given-names>
</name>
<etal/>
</person-group> (<year>2022b</year>). <article-title>Chain-of-thought prompting elicits reasoning in large language models</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2201.11903">https://arxiv.org/abs/2201.11903</ext-link>
</comment>.</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wells</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Dantam</surname>
<given-names>N. T.</given-names>
</name>
<name>
<surname>Shrivastava</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kavraki</surname>
<given-names>L. E.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Learning feasibility for task and motion planning in tabletop environments</article-title>. <source>IEEE Ral.</source> <volume>4</volume>, <fpage>1255</fpage>&#x2013;<lpage>1262</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2019.2894861</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weng</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Ll-powered autonomous agents</article-title>. <source>lilianweng.github.Io.</source>
</citation>
</ref>
<ref id="B55">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>White</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Hays</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sandborn</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Olea</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gilbert</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>A prompt pattern catalog to enhance prompt engineering with chatgpt</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2302.11382">https://arxiv.org/abs/2302.11382</ext-link>
</comment>.</citation>
</ref>
<ref id="B56">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Wolf</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Debut</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sanh</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Chaumond</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Delangue</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Moi</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Huggingface&#x2019;s transformers: state-of-the-art natural language processing</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1910.03771">https://arxiv.org/abs/1910.03771</ext-link>
</comment>.</citation>
</ref>
<ref id="B57">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chalvatzaki</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Accelerating integrated task and motion planning with neural feasibility checking</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2203.10568">https://arxiv.org/abs/2203.10568</ext-link>
</comment>.</citation>
</ref>
<ref id="B58">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Welker</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Choromanski</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tombari</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Purohit</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Socratic models: composing zero-shot multimodal reasoning with language</article-title>. <comment>Available at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/2204.00598">https://arxiv.org/abs/2204.00598</ext-link>
</comment>.</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Cui</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Evaluating commonsense in pre-trained language models</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>34</volume>, <fpage>9733</fpage>&#x2013;<lpage>9740</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v34i05.6523</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>