<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="review-article" dtd-version="1.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title-group>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1606247</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1606247</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Review</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Diffusion models for robotic manipulation: a survey</article-title>
<alt-title alt-title-type="left-running-head">Wolf et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1606247">10.3389/frobt.2025.1606247</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Wolf</surname>
<given-names>Rosa</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3026130"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Shi</surname>
<given-names>Yitian</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<uri xlink:href="https://loop.frontiersin.org/people/3165106"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Sheng</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<uri xlink:href="https://loop.frontiersin.org/people/3026441"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Rayyes</surname>
<given-names>Rania</given-names>
</name>
<xref ref-type="aff" rid="aff1"/>
<uri xlink:href="https://loop.frontiersin.org/people/493962"/>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &amp; editing</role>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>AI and Robotics (AIR), Institute of Material Handling and Logistics (IFL), Karlsruhe Institute of Technology (KIT)</institution>, <city>Karlsruhe</city>, <country country="DE">Germany</country>
</aff>
<author-notes>
<corresp id="c001">
<label>&#x2a;</label>Correspondence: Rosa Wolf, <email xlink:href="rosa.wolf@kit.edu">rosa.wolf@kit.edu</email>
</corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-09-09">
<day>09</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1606247</elocation-id>
<history>
<date date-type="received">
<day>04</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Wolf, Shi, Liu and Rayyes.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Wolf, Shi, Liu and Rayyes</copyright-holder>
<license>
<ali:license_ref start_date="2025-09-09">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<p>Diffusion generative models have demonstrated remarkable success in visual domains such as image and video generation. They have also recently emerged as a promising approach in robotics, especially in robot manipulations. Diffusion models leverage a probabilistic framework, and they stand out with their ability to model multi-modal distributions and their robustness to high-dimensional input and output spaces. This survey provides a comprehensive review of state-of-the-art diffusion models in robotic manipulation, including grasp learning, trajectory planning, and data augmentation. Diffusion models for scene and image augmentation lie at the intersection of robotics and computer vision for vision-based tasks to enhance generalizability and data scarcity. This paper also presents the two main frameworks of diffusion models and their integration with imitation learning and reinforcement learning. In addition, it discusses the common architectures and benchmarks and points out the challenges and advantages of current state-of-the-art diffusion-based methods.</p>
</abstract>
<kwd-group>
<kwd>diffusion models</kwd>
<kwd>robot manipulation learning</kwd>
<kwd>generative models</kwd>
<kwd>imitation learning</kwd>
<kwd>grasp learning</kwd>
</kwd-group>
<funding-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. Funded by the Deutsche Forschungsgemeinschaft (DFG, German Research Foundation) &#x2013; SFB-1574 &#x2013; 471687386.</funding-statement>
</funding-group>
<counts>
<fig-count count="2"/>
<table-count count="7"/>
<equation-count count="9"/>
<ref-count count="208"/>
<page-count count="25"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Robot Learning and Evolution</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Diffusion Models (DMs) have emerged as highly promising deep generative models in diverse domains, including computer vision (<xref ref-type="bibr" rid="B45">Ho et al., 2020</xref>; <xref ref-type="bibr" rid="B152">Song J. et al., 2021</xref>; <xref ref-type="bibr" rid="B111">Nichol and Dhariwal, 2021</xref>; <xref ref-type="bibr" rid="B127">Ramesh et al., 2022</xref>; <xref ref-type="bibr" rid="B132">Rombach et al., 2022a</xref>), natural language processing (<xref ref-type="bibr" rid="B77">Li et al., 2022</xref>; <xref ref-type="bibr" rid="B199">Zhang et al., 2023</xref>; <xref ref-type="bibr" rid="B185">Yu et al., 2022</xref>), and robotics (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>; <xref ref-type="bibr" rid="B162">Urain et al., 2023</xref>). DMs intrinsically posses the ability to model any distribution. They have demonstrated remarkable performance and stability in modeling complex and multi-modal distributions<xref ref-type="fn" rid="n1">
<sup>1</sup>
</xref> from high-dimensional and visual data surpassing the ability of Gaussian Mixture Models (GMMs) or Energy-based models (EBMs) like Implicit behavior cloning (IBC) (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>). While GMMs and IBCs can model multi-modal distributions, and IBCs can even learn complex discontinuous distributions (<xref ref-type="bibr" rid="B32">Florence et al., 2022</xref>), experiments (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>) show that in practice, they might be heavily biased toward specific modes. In general, DMs have also demonstrated performance exceeding generative adversarial networks (GANs) (<xref ref-type="bibr" rid="B70">Krichen, 2023</xref>), which were previously considered the leading paradigm in the field of generative models. GANs usually require adversarial training, which can lead to mode collapse and training instability (<xref ref-type="bibr" rid="B70">Krichen, 2023</xref>). Additionally, GANs have been reported to be sensitive to hyperparameters (<xref ref-type="bibr" rid="B90">Lucic et al., 2018</xref>). </p>
<p>Since 2022, there has been a noticeable increase in the implementation of diffusion probabilistic models within the field of robotic manipulation. These models are applied across various tasks, including trajectory planning, e.g. (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>), and grasp prediction, e.g., (<xref ref-type="bibr" rid="B162">Urain et al., 2023</xref>). The ability of DMs to model multi-modal distributions is a great advantage in many robotic manipulation applications. In various manipulation tasks, such as trajectory planning and grasping, there exist multiple equally valid solutions (redundant solutions). Capturing all solutions improves generalizability and robots&#x2019; versatility, as it enables generating feasible solutions under different conditions, such as different placements of objects or different constraints during inference. Although in the context of trajectory planning using DMs, primarily imitation learning is applied, DMs have been adapted for integration with reinforcement learning (RL), e.g., (<xref ref-type="bibr" rid="B36">Geng et al., 2023</xref>). Research efforts focus on various components of the diffusion process adapted to different tasks in the domain of robotic manipulation. To give just some examples, developed architectures integrate different or even multiple input modalities. One example of an input modality could be point clouds (<xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>). With the provided depth information, models can learn more complex tasks, for which a better 3D scene understanding is crucial. Another example of an additional input modality could be natural language (<xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>; <xref ref-type="bibr" rid="B28">Du et al., 2023</xref>; <xref ref-type="bibr" rid="B72">Li et al., 2025</xref>), which also enables the integration of foundation models, like large language models, into the workflow. In <xref ref-type="bibr" rid="B190">Ze et al. (2024)</xref>, both point clouds and language task instructions are used as multiple input modalities. Others integrate DMs into hierarchical planning (<xref ref-type="bibr" rid="B92">Ma X. et al., 2024</xref>; <xref ref-type="bibr" rid="B28">Du et al., 2023</xref>) or skill learning (<xref ref-type="bibr" rid="B80">Liang et al., 2024</xref>; <xref ref-type="bibr" rid="B104">Mishra et al., 2023</xref>), to facilitate their state-of-the-art capabilities in modeling high-dimensional data and multi-modal distributions, for long-horizon and multi-task settings. Many methodologies, e.g., (<xref ref-type="bibr" rid="B62">Kasahara et al., 2024</xref>; <xref ref-type="bibr" rid="B22">Chen Z. et al., 2023</xref>), employ diffusion-based data augmentation in vision-based manipulation tasks to scale up datasets and reconstruct scenes. It is important to note that one of the major challenges of DMs is its comparatively slow sampling process, which has been addressed in many methods, e.g., (<xref ref-type="bibr" rid="B152">Song J. et al., 2021</xref>; <xref ref-type="bibr" rid="B19">Chen K. et al., 2024</xref>; <xref ref-type="bibr" rid="B206">Zhou H. et al., 2024</xref>), also enabling real-time prediction.</p>
<p>To the best of our knowledge, we provide the first survey of DMs concentrating on the field of robotic manipulation. The survey offers a systematic classification of various methodologies related to DMs within the realm of robotic manipulation, regarding network architecture, learning framework, application, and evaluation. Alongside comprehensive descriptions, we present illustrative taxonomies.</p>
<p>To provide the reader with the necessary background information on DMs, we will first introduce their fundamental mathematical concepts (<xref ref-type="sec" rid="s2">Section 2</xref>). This section provides a general overview of DMs rather than focusing specifically on robotic manipulation. Then, network architectures commonly used for DMs in robotic manipulation will be discussed (<xref ref-type="sec" rid="s3">Section 3</xref>). Next (<xref ref-type="sec" rid="s4">Section 4</xref>), we explore the three primary applications of DMs in robotic manipulation: trajectory generation (<xref ref-type="sec" rid="s4-1">Section 4.1</xref>), robotic grasp synthesis (<xref ref-type="sec" rid="s4-2">Section 4.2</xref>), and visual data augmentation (<xref ref-type="sec" rid="s4-3">Section 4.3</xref>). This is followed by an overview of commonly used benchmarks and baselines (<xref ref-type="sec" rid="s5">Section 5</xref>). Finally, we discuss our conclusions and existing limitations, and outline potential directions for future research (<xref ref-type="sec" rid="s6">Section 6</xref>).</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Preliminaries on diffusion models</title>
<sec id="s2-1">
<label>2.1</label>
<title>Mathematical framework</title>
<p>The key idea of DMs is to gradually perturb an unknown target distribution <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>data</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> into a simple known distribution, e.g., a normal Gaussian distribution, which is first introduced in (<xref ref-type="bibr" rid="B151">Sohl-Dickstein et al., 2015</xref>). To generate new data, points are sampled from the initial known &#x201c;simple&#x201d; distribution, and perturbations are estimated to iteratively reverse the diffusion process. The forward and backward diffusion processes are also visualized in <xref ref-type="fig" rid="F1">Figure 1</xref>. There exist two main approaches to diffusion-based modeling, both based on the original work by <xref ref-type="bibr" rid="B151">Sohl-Dickstein et al. (2015)</xref>. The first group of methods is score-based DMs, where the gradient of the log-likelihood of the data is learned to reverse the diffusion process. This score-based generative modeling was first introduced in <xref ref-type="bibr" rid="B154">Song and Ermon (2019)</xref>. In the other group of methods, a network is trained to directly predict the noise, which is added during the forward process. This methodology was first introduced in Denoising Diffusion Probabilistic Models (DDPM) (<xref ref-type="bibr" rid="B45">Ho et al., 2020</xref>).</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Illustrations of diffusion (forward) processes on image, trajectories, and grasp poses (<xref ref-type="bibr" rid="B162">Urain et al., 2023</xref>) and their corresponding synthesis (backward) processes.</p>
</caption>
<graphic xlink:href="frobt-12-1606247-g001.tif">
<alt-text content-type="machine-generated">Forward (diffusion) and backward (synthesis) processes for image synthesis, robot trajectory synthesis, and grasp synthesis. In the top row, images are shown. In the images, a clear number &#x201c;8&#x201d; on the left becomes progressively noisier to the right until the points are evenly spread in a circle. In the middle row, robot trajectories are shown as sequences of points leading toward a goal, marked with coin symbols, while avoiding obstacles indicated by traffic cones. Moving right, the trajectories become progressively noisier until they are distributed according to white noise. In the bottom row, a teapot is shown with different possible grasps. On the left, grasps are precise and clustered; as noise increases to the right, they spread evenly. Noise levels are labeled with \(k\) values. Diffusion goes left to right, synthesis right to left.</alt-text>
</graphic>
</fig>
<p>The original score-based DM by <xref ref-type="bibr" rid="B154">Song and Ermon (2019)</xref> is rarely used in the field of robotic manipulation. This could be due to its inefficient sampling process. However, as it forms a crucial mathematical framework and baseline for many of the later developed DMs, e.g., (<xref ref-type="bibr" rid="B155">Song Y. et al., 2021</xref>; <xref ref-type="bibr" rid="B61">Karras et al., 2022</xref>), including DDPM <xref ref-type="bibr" rid="B45">Ho et al. (2020)</xref>, we describe the main concepts in the following section. While DDPM is rarely used as well, the commonly used method Denoising Diffusion Implicit Models (DDIM) (<xref ref-type="bibr" rid="B152">Song J. et al., 2021</xref>) originates from DDPM. DDIM only alters the sampling process of DDPM while keeping its training procedure. Hence, understanding DDPM is crucial for many applications of DMs in robotic manipulation.</p>
<p>In the following sections, we first introduce score-based DMs, then DDPM, before addressing their shortcomings.</p>
<sec id="s2-1-1">
<label>2.1.1</label>
<title>Denoising score matching using Noise Conditional Score Networks</title>
<p>One approach to estimate perturbations in the data distribution is to use denoising score matching with Lagenvin dynamics (SMLD), where the score of the data density of the perturbed distributions is learned using a Noise Conditional Score Network (NCSM) (<xref ref-type="bibr" rid="B154">Song and Ermon, 2019</xref>). This method is described in this section, and for more details, please refer to their original work. During the forward diffusion process, data <inline-formula id="inf2">
<mml:math id="m2">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> from an unknown distribution <inline-formula id="inf3">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>data</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is transformed into random noise <inline-formula id="inf4">
<mml:math id="m4">
<mml:mrow>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, by gradually adding noise. New data is generated during the reverse process, where the learned NCSM is used to iteratively denoise the initial samples.</p>
<sec id="s2-1-1-1">
<label>2.1.1.1</label>
<title>Forward process</title>
<p>Let <inline-formula id="inf5">
<mml:math id="m5">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> be a noise schedule with progressively increasing variance, i.e., <inline-formula id="inf6">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> for all <inline-formula id="inf7">
<mml:math id="m7">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. To get from the true data distribution <inline-formula id="inf8">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>data</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to the perturbed data distribution <inline-formula id="inf9">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, with variance <inline-formula id="inf10">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, noise is added to the data according to a pre-specified noise distribution <inline-formula id="inf11">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. To denoise the data, the gradients of the logarithmic probability density functions <inline-formula id="inf12">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>log</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., the scores, are estimated using the NCSM. To train the NCSM <inline-formula id="inf13">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, for all noise scales <inline-formula id="inf14">
<mml:math id="m14">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> the weighted sum of denoising score matchings is minimized (<xref ref-type="bibr" rid="B154">Song and Ermon, 2019</xref>):<disp-formula id="e1">
<mml:math id="m15">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>data</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-1-1-2">
<label>2.1.1.2</label>
<title>Reverse process</title>
<p>Starting with randomly drawn noise samples <inline-formula id="inf15">
<mml:math id="m16">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, Langevin dynamics are applied recursively over all <inline-formula id="inf16">
<mml:math id="m17">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, to generate samples using the learned score function:<disp-formula id="e2">
<mml:math id="m18">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mi>n</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced open="{" close="}">
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>.</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>where <inline-formula id="inf17">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3e;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> is the step size and <inline-formula id="inf18">
<mml:math id="m20">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is randomly drawn noise. During one Langevin dynamic for noise scale <inline-formula id="inf19">
<mml:math id="m21">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the index <inline-formula id="inf20">
<mml:math id="m22">
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is increasing until <inline-formula id="inf21">
<mml:math id="m23">
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Then, the final value <inline-formula id="inf22">
<mml:math id="m24">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, of one Langevin dynamic becomes the initial value <inline-formula id="inf23">
<mml:math id="m25">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> for the next Langevin dynamic with the next lower noise scale <inline-formula id="inf24">
<mml:math id="m26">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, i.e., <inline-formula id="inf25">
<mml:math id="m27">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>. For small enough step sizes, the final generated samples <inline-formula id="inf26">
<mml:math id="m28">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, should be approximately distributed according to <inline-formula id="inf27">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>data</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
<sec id="s2-1-2">
<label>2.1.2</label>
<title>Denoising Diffusion Probabilistic Models (DDPM)</title>
<p>In DDPM (<xref ref-type="bibr" rid="B45">Ho et al., 2020</xref>), instead of estimating the score function directly, a noise prediction network, conditioned on the noise scale, is trained. Similarly to SMLD with NCSN, new points are generated by sampling Gaussian noise and iteratively denoising the samples using the learned noise prediction network.</p>
<p>Notably, there is one step per noise scale in the denoising process instead of recursively sampling from each noise scale.</p>
<sec id="s2-1-2-1">
<label>2.1.2.1</label>
<title>Forward process</title>
<p>To train the noise prediction network <inline-formula id="inf28">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, first points <inline-formula id="inf29">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold">0</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>data</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are sampled from the true unknown data distribution. The samples are degraded by adding noise <inline-formula id="inf30">
<mml:math id="m32">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> until at degrading step <inline-formula id="inf31">
<mml:math id="m33">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the degraded samples are approximately normally distributed, i.e. <inline-formula id="inf32">
<mml:math id="m34">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. As already introduced by <xref ref-type="bibr" rid="B151">Sohl-Dickstein et al. (2015)</xref>, the noise is added according to a Markovian process:<disp-formula id="e3">
<mml:math id="m35">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="normal">I</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>where <inline-formula id="inf33">
<mml:math id="m36">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0,1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the noise variance schedule, which can either be a hyperparameter (<xref ref-type="bibr" rid="B45">Ho et al., 2020</xref>), or optimized as part of the model training process (<xref ref-type="bibr" rid="B111">Nichol and Dhariwal, 2021</xref>). In practice, instead of adding noise iteratively, the formulation also allows adding the noise in closed form:<disp-formula id="e4">
<mml:math id="m37">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mi mathvariant="normal">I</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>with <inline-formula id="inf34">
<mml:math id="m38">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2254;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo movablelimits="false" form="prefix">&#x220f;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf35">
<mml:math id="m39">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2254;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. This allows first uniformly sampling a noise scale <inline-formula id="inf36">
<mml:math id="m40">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">U</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and then directly inferring the corresponding degraded sample.</p>
<p>Adding the noise in closed form facilitates training a noise prediction network <inline-formula id="inf37">
<mml:math id="m41">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> by minimizing the mean squared error for <inline-formula id="inf38">
<mml:math id="m42">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>:<disp-formula id="e5">
<mml:math id="m43">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mfenced open="|" close="|">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-1-2-2">
<label>2.1.2.2</label>
<title>Reverse process</title>
<p>Similar to the reverse process described in <xref ref-type="sec" rid="s2-1-1">Section 2.1.1</xref>, new samples are generated from random noise <inline-formula id="inf39">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">I</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, using the learned forward process <inline-formula id="inf40">
<mml:math id="m45">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. As the forward process is modeled using Gaussian distributions, the reverse process <inline-formula id="inf41">
<mml:math id="m46">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is also a Gaussian distribution if the number of diffusion steps is sufficiently large, i.e., the step size is small enough (<xref ref-type="bibr" rid="B151">Sohl-Dickstein et al., 2015</xref>):<disp-formula id="e6">
<mml:math id="m47">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2248;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>In DDPM, the variance-schedule is fixed and thus <inline-formula id="inf42">
<mml:math id="m48">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="normal">I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. Additionally, using reparameterization, it can be shown that the mean of the distribution at each step can be iteratively predicted using the previous value <inline-formula id="inf43">
<mml:math id="m49">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the estimated noise <inline-formula id="inf44">
<mml:math id="m50">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B45">Ho et al., 2020</xref>):<disp-formula id="e7">
<mml:math id="m51">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msqrt>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msqrt>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="bold">z</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>which is repeated until <inline-formula id="inf45">
<mml:math id="m52">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is computed. As in SMLD, for small enough step sizes, the final generated samples <inline-formula id="inf46">
<mml:math id="m53">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are approximately distributed according to the true data distribution <inline-formula id="inf47">
<mml:math id="m54">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>data</mml:mtext>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</sec>
</sec>
</sec>
<sec id="s2-2">
<label>2.2</label>
<title>Architectural improvements and adaptations</title>
<p>One of the main disadvantages of DMs is the iterative sampling, leading to a relatively slow sampling process. In comparison, using GANs or variational autoencoders (VAEs), only a single forward pass through the trained network is required to produce a sample. In both DDPM and the original formulation of SMLD, the number of time steps (noise levels) in the forward and reverse processes is equal. While reducing the number of noise levels leads to a faster sampling process, it comes at the cost of sample quality. Thus, there have been numerous works to adapt the architectures and sampling processes of DDPM and SMLD to improve both the sampling speed and quality of DMs, e.g., (<xref ref-type="bibr" rid="B111">Nichol and Dhariwal, 2021</xref>; <xref ref-type="bibr" rid="B152">Song J. et al., 2021</xref>; <xref ref-type="bibr" rid="B155">Song Y. et al., 2021</xref>).</p>
<sec id="s2-2-1">
<label>2.2.1</label>
<title>Improving sampling speed and quality</title>
<p>The forward diffusion process can be formulated as a stochastic differential equation (SDE). Using the corresponding reverse-time SDE, SDE-solvers can then be applied to generate new samples (<xref ref-type="bibr" rid="B155">Song Y. et al., 2021</xref>). <xref ref-type="bibr" rid="B155">Song et al. (2021b)</xref> shows that the diffusion process from SMLD corresponds to an SDE where the variance of the perturbation kernels <inline-formula id="inf48">
<mml:math id="m55">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is exploding with increasing <inline-formula id="inf49">
<mml:math id="m56">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. This is referred to as the variance exploding SDE (VE SDE) in the literature. The diffusion process from DDPM corresponds to a variance-preserving SDE, referred to as VP SDE in the literature. As such, the original formulations of SMLD and DDPM can be interpreted as specific discretizations of their corresponding SDEs. <xref ref-type="bibr" rid="B155">Song Y. et al. (2021)</xref> also shows that once the score-network is trained, the reverse-time SDE can be replaced by an ordinary differential equation (ODE). Using an ODE has several advantages. As the reverse process is deterministic, it allows for precise likelihood computation (<xref ref-type="bibr" rid="B155">Song Y. et al., 2021</xref>). Moreover, the deterministic process naturally leads to higher consistency. Thus, the ODE formulation can be used as a high-level feature-preserving encoding, which also allows interpolations in latent space (<xref ref-type="bibr" rid="B152">Song J. et al., 2021</xref>; <xref ref-type="bibr" rid="B61">Karras et al., 2022</xref>). Finally, using ODEs enables faster and adaptive sampling, which is why it forms the baseline for many of the following methods.</p>
<p>One group of methods aimed at improving sampling speed (<xref ref-type="bibr" rid="B57">Jolicoeur-Martineau et al., 2021</xref>; <xref ref-type="bibr" rid="B152">Song J. et al., 2021</xref>; <xref ref-type="bibr" rid="B88">Lu et al., 2022</xref>; <xref ref-type="bibr" rid="B61">Karras et al., 2022</xref>) designs samplers that operate independently of the specific training process. Using an SDE/ODE-based formulation allows choosing different discretizations of the reverse process than for the forward process. Larger step sizes reduce computational cost and sampling time but introduce greater truncation error. The sampler operates independently of the specific noise prediction network implementation, enabling the use of a single network, such as one trained with DDPM, with different samplers.</p>
<p>Denoising Diffusion Implicit Models (DDIM) (<xref ref-type="bibr" rid="B111">Nichol and Dhariwal, 2021</xref>) is the dominant method used for robotic manipulation. It uses a deterministic sampling process and outperforms DDPM when using only a few (10&#x2013;100) sampling iterations. DDIM can be formulated as a first-order ODE solver. In Diffusion Probabilistic Models-solver (DPM-solver) (<xref ref-type="bibr" rid="B88">Lu et al., 2022</xref>), a second-order ODE solver is applied, which decreases the truncation error, thus further increasing performance on several image classification benchmarks for a low number of sampling steps. In contrast to DDIM, <xref ref-type="bibr" rid="B61">Karras et al. (2022)</xref>; <xref ref-type="bibr" rid="B88">Lu et al. (2022)</xref> use non-uniform step sizes in the solver. In a detailed analysis <xref ref-type="bibr" rid="B61">Karras et al. (2022)</xref> empirically shows that compared to uniform step-sizes, linear decreasing step sizes during denoising lead to increased performance (<xref ref-type="bibr" rid="B61">Karras et al., 2022</xref>), indicating that errors near the true distribution have a larger impact.</p>
<p>Even though DPM-solver (<xref ref-type="bibr" rid="B88">Lu et al., 2022</xref>) shows superior performance over DDIM. It should be noted that in the original papers (<xref ref-type="bibr" rid="B152">Song J. et al., 2021</xref>; <xref ref-type="bibr" rid="B88">Lu et al., 2022</xref>), only image-classification benchmarks are considered to compare both methods. Therefore, more extensive tests should be performed to validate these results.</p>
<p>A second group of methods addressing sampling speed also adapts the training process or requires additional fine-tuning. Examples are knowledge distillation of DMs to gradually reduce the number of noise levels (<xref ref-type="bibr" rid="B141">Salimans and Ho, 2022</xref>), or finetuning of the noise schedule (<xref ref-type="bibr" rid="B111">Nichol and Dhariwal, 2021</xref>; <xref ref-type="bibr" rid="B171">Watson et al., 2022</xref>). While in DDPM and DDIM, the noise schedule is fixed, in improved Denoising Diffusion Probabilistic Models (iDDPM) (<xref ref-type="bibr" rid="B111">Nichol and Dhariwal, 2021</xref>), the noise schedule is learned, resulting in better sample quality. They also suggest changing from a linear noise schedule, like in DDPM, to other schedules, e.g., a cosine noise schedule. In particular, for low-resolution samples, a linear schedule leads to a noisy diffusion process with too rapid information loss, while the cosine noise schedule has smaller steps during the beginning and end of the diffusion process. Already after a fraction of around 0.6 diffusion steps, the linear noise schedule is close to zero (and the data distribution close to white noise). Thus, the first steps of the reverse process do not strongly contribute to the data generation process, making the sampling process inefficient. Although iDDPM (<xref ref-type="bibr" rid="B111">Nichol and Dhariwal, 2021</xref>) also outperforms DDIM, it requires fine-tuning, which might be a reason why it is less popular.</p>
<p>There are also several methods (<xref ref-type="bibr" rid="B206">Zhou H. et al., 2024</xref>; <xref ref-type="bibr" rid="B76">Li X. et al., 2024</xref>; <xref ref-type="bibr" rid="B170">Wang et al., 2023b</xref>; <xref ref-type="bibr" rid="B19">Chen K. et al., 2024</xref>) regarding sampling speed, specifically for applications in robotic manipulation, which is different from the previously named methodologies, which were developed in the context of image processing. For example, <xref ref-type="bibr" rid="B19">Chen K. et al. (2024)</xref> samples from a more informed distribution than a Gaussian. They point out that even initial distributions approximated with simple heuristics result in better sample quality, especially when using few diffusion steps or when only a limited amount of data is available. Others (<xref ref-type="bibr" rid="B122">Prasad et al., 2024</xref>) use teacher&#x2013;student distillation techniques (<xref ref-type="bibr" rid="B157">Tarvainen and Valpola, 2017</xref>), where pretrained diffusion models serve as teachers, guiding student models to operate with larger denoising steps while preserving consistency with the teacher&#x2019;s results at smaller steps. While this increases training effort, it decreases sampling time at inference, which is especially important in (near) real-time control.</p>
<p>Recently, flow matching (<xref ref-type="bibr" rid="B82">Lipman et al., 2023</xref>) has been used as an alternative method to diffusion. Like with diffusion, the true distribution is estimated starting from a noise distribution. However, instead of learning the time-dependent score or noise, and then deriving the velocity from noise to data distribution from it, in flow matching, the time-dependent velocity field is learned directly. This leads to a simpler training objective, using the interpolation between the noise sample and true data point, without requiring a noise schedule. Thus, flow matching is usually more numerically stable and requires less hyperparameter tuning. However, when using few sampling steps, with flow matching, there is a risk of mode-collapse and infeasible solutions, as the ODE-solver averages over the velocity field. Thus, <xref ref-type="bibr" rid="B33">Frans et al. (2025)</xref> conditions the model not only on the time-step, but also on the step-size. By using the fact that one large step should lead to the same point as two consecutive steps of half the size, they maximize a self-consistency objective in addition to the flow-matching objective. Thus, the model can sample with a single step, with only a small drop in performance, far surpassing the performance of DDIM, when only a small number of sampling steps are used. While this is similar to the above-mentioned distillation techniques (<xref ref-type="bibr" rid="B122">Prasad et al., 2024</xref>), here only a single model has to be trained.</p>
</sec>
</sec>
<sec id="s2-3">
<label>2.3</label>
<title>Adaptations for robotic manipulation</title>
<p>Two main points must be considered to apply DMs to robotic manipulation. Firstly, in the diffusion processes described in the previous sections, given the initial noise, samples are generated solely based on the trained noise prediction network or conditional score network. However, robot actions are usually dependent on simulated or real-world observations with multi-modal sensory data and the robot&#x2019;s proprioception. Thus, the network used in the denoising process has to be conditioned on these observations (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>). Encoding observations varies in different algorithms. Some use ground truth state information, such as object positions (<xref ref-type="bibr" rid="B1">Ada et al., 2024</xref>), and object features, like object sizes (<xref ref-type="bibr" rid="B104">Mishra et al., 2023</xref>; <xref ref-type="bibr" rid="B99">Mendez-Mendez et al., 2023</xref>). In this case, sim-to-real transfer is challenging due to sensor inaccuracies, object occlusions, or other adversarial settings, e.g., lightning conditions, Therefore, most methods directly condition on visual observations, such as images (<xref ref-type="bibr" rid="B148">Si et al., 2024</xref>; <xref ref-type="bibr" rid="B5">Bharadhwaj et al., 2024a</xref>; <xref ref-type="bibr" rid="B164">Vosylius et al., 2024</xref>; <xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>; <xref ref-type="bibr" rid="B145">Shi et al., 2023</xref>), point clouds (<xref ref-type="bibr" rid="B87">Liu et al., 2023c</xref>; <xref ref-type="bibr" rid="B72">Li et al., 2025</xref>), or feature encodings and embeddings (<xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>; <xref ref-type="bibr" rid="B76">Li X. et al., 2024</xref>; <xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>; <xref ref-type="bibr" rid="B80">Liang et al., 2024</xref>; <xref ref-type="bibr" rid="B179">Xian et al., 2023</xref>; <xref ref-type="bibr" rid="B180">Xu et al., 2023</xref>), where the robustness to adversarial setting can be directly addressed.</p>
<p>Secondly, unlike in image generation, where the pixels are spatially correlated, in trajectory generation for robotic manipulation, the samples of a trajectory are temporally correlated. On the one hand, generating complete trajectories may not only lead to high inaccuracies and error accumulation of the long-horizon predictions, but also prevent the model from reacting to changes in the environment. On the other hand, predicting the trajectory one action at a time increases the compounding error effect and may lead to frequent switches between modes. Accordingly, trajectories are mostly predicted in subsequences, with a receding horizon, e.g., (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>; <xref ref-type="bibr" rid="B142">Scheikl et al., 2024</xref>), which will be discussed in more detail in <xref ref-type="sec" rid="s4-1">Section 4.1</xref> and is visualized in <xref ref-type="fig" rid="F2">Figure 2</xref>. In receding horizon control, the diffusion model generates only a subtrajectory with each backward pass. The subtrajectory is executed before generating the next subtrajectory on the updated observations. In comparison, grasps are generated similarly to images. As here only a single action, usually the grasp pose, is generated, this is done using a single backward pass of the diffusion model. Moreover, the grasp pose is usually predicted from a single initial observation. During execution, possible changes in the scene are not being taken into account. The backward pass for generating one action is visualized in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Illustrations of the iterative trajectory generation using receding horizon control. At inference, the trajectory is planned up to a planning horizon <inline-formula id="inf50">
<mml:math id="m57">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, conditioned on the past <inline-formula id="inf51">
<mml:math id="m58">
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> observations <inline-formula id="inf52">
<mml:math id="m59">
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Of this plan, only the steps until the control horizon <inline-formula id="inf53">
<mml:math id="m60">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> are executed. In the figure, this is visualized in the outer loop with the time variable <inline-formula id="inf54">
<mml:math id="m61">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>. In the inner denoising loop, one subtrajectory <inline-formula id="inf55">
<mml:math id="m62">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> at the current time step <inline-formula id="inf56">
<mml:math id="m63">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is generated, using a diffusion model. Conditioned on the last <inline-formula id="inf57">
<mml:math id="m64">
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> observations and the current noise level <inline-formula id="inf58">
<mml:math id="m65">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the diffusion model predicts the noise, or score, dependent on the model type. Using the predicted noise/score, the trajectory at the next lower noise level <inline-formula id="inf59">
<mml:math id="m66">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> is calculated. This is then used as the next input to the diffusion model until the trajectory is completely denoised <inline-formula id="inf60">
<mml:math id="m67">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, at which point it is executed. After execution of the subtrajectory, the time is increased and the next <inline-formula id="inf61">
<mml:math id="m68">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> steps of the trajectory are planned. For training, ground truth trajectories and corresponding observations are sampled from the data buffer. The diffusion model is also trained on subtrajectories. However, the lookahead <inline-formula id="inf62">
<mml:math id="m69">
<mml:mrow>
<mml:mi>H</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> during training may be chosen larger than during inference, to ensure flexibility. The diffusion model is trained to predict the noise of a noisy trajectory. For this, first, a noise level <inline-formula id="inf63">
<mml:math id="m70">
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is sampled. Then the noise <inline-formula id="inf64">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is sampled, according to the predefined variance schedule. The noise is added in closed form to the ground-truth trajectory <inline-formula id="inf65">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> (see <xref ref-type="disp-formula" rid="e4">Equation 4</xref>) to get the noisy trajectory <inline-formula id="inf66">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The predicted noise <inline-formula id="inf67">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> on the trajectory <inline-formula id="inf68">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is compared with the true sampled noise <inline-formula id="inf69">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> to compute the loss. Using this, the diffusion model can be updated.</p>
</caption>
<graphic xlink:href="frobt-12-1606247-g002.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a diffusion model for trajectory generation. The training phase involves a data buffer, containing observation frames, and ground truth trajectories, sampling noise levels and the according noise, adding the noise to the ground truth trajectory and predicting the noise with a noise/score prediction network, which is updated based on a loss function. Inference includes a denoising loop, and a execution phase where a robotic arm executes the next part of the denoised trajectory sequence.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Architecture</title>
<sec id="s3-1">
<label>3.1</label>
<title>Network architecture</title>
<p>For the implementation of the DM, it is essential to select an appropriate architecture for the noise prediction network. There exist three predominant architectures used for the denoising diffusion networks: Convolutional neural networks (CNNs), transformers, and Multi-Layer Perceptrons (MLPs).</p>
<sec id="s3-1-1">
<label>3.1.1</label>
<title>Convolutional neural networks</title>
<p>The most frequently employed architecture is the CNN, more specifically the Temporal U-Net that was first introduced by <xref ref-type="bibr" rid="B55">Janner et al. (2022)</xref> in their algorithm Diffuser, a DM for robotics tasks. The U-Net architecture (<xref ref-type="bibr" rid="B135">Ronneberger et al., 2015</xref>) has shown great success in image generation with DMs, e.g., (<xref ref-type="bibr" rid="B45">Ho et al., 2020</xref>; <xref ref-type="bibr" rid="B24">Dhariwal and Nichol, 2021</xref>; <xref ref-type="bibr" rid="B155">Song Y. et al., 2021</xref>). U-net, in general, is proven to be sample efficient and can even generalize well with small training datasets (<xref ref-type="bibr" rid="B101">Meyer-Veit et al., 2022b</xref>; <xref ref-type="bibr" rid="B100">Meyer-Veit et al., 2022a</xref>). Thus, it has been adapted to robotic manipulation by replacing two-dimensional spatial convolutions with one-dimensional temporal convolutions (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>).</p>
<p>The temporal U-Net is further adapted by <xref ref-type="bibr" rid="B23">Chi et al. (2023)</xref> in their CNN-based Diffusion Policy (DP) for robotic manipulation. While in Diffuser, the state and action trajectories are jointly denoised, only the action trajectories are generated in DP. To ensure temporal consistency, the diffusion process is conditioned on a history of observations using feature-wise linear modification (FiLM) (<xref ref-type="bibr" rid="B118">Perez et al., 2018</xref>). This formulation allows for an extension to different and multiple conditions by concatenating them in feature space before applying FiLM (<xref ref-type="bibr" rid="B76">Li X. et al., 2024</xref>; <xref ref-type="bibr" rid="B148">Si et al., 2024</xref>; <xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B72">Li et al., 2025</xref>; <xref ref-type="bibr" rid="B167">Wang L. et al., 2024</xref>). Moreover, it also enables the incorporation of constraints embedded with an MLP (<xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>; <xref ref-type="bibr" rid="B208">Zhou et al., 2023</xref>; <xref ref-type="bibr" rid="B121">Power et al., 2023</xref>).</p>
<p>Discussed in more detail in <xref ref-type="sec" rid="s4-1-1-6">Section 4.1.1.6</xref>, <xref ref-type="bibr" rid="B55">Janner et al. (2022)</xref> formulates conditioning as inpainting, where during inferences at each denoising step, specific states from the currently being generated sample are replaced with states from the condition. For example, the final state of a generated trajectory may be replaced by the goal state, for goal-conditioning. This only affects the sampling process at inference and, thus, does not require any adaptations of the network architecture. However, it only supports point-wise conditions, severely limiting its applications. Multiple frameworks (<xref ref-type="bibr" rid="B140">Saha et al., 2024</xref>; <xref ref-type="bibr" rid="B15">Carvalho et al., 2023</xref>; <xref ref-type="bibr" rid="B170">Wang et al., 2023b</xref>; <xref ref-type="bibr" rid="B92">Ma X. et al., 2024</xref>) directly employ the temporal U-Net architecture introduced by <xref ref-type="bibr" rid="B55">Janner et al. (2022)</xref>. However, as this type of conditioning is highly limited in its applications, FiLM conditioning is more common. A different but less-used architecture incorporates conditions via cross-attention mapped to the intermediate layers of the U-Net (<xref ref-type="bibr" rid="B192">Zhang E. et al., 2024</xref>), which is more complicated to integrate than FiLM conditioning.</p>
</sec>
<sec id="s3-1-2">
<label>3.1.2</label>
<title>Transformers</title>
<p>Another commonly used architecture for the denoising network are transformers. A history of observations, the current denoising time step, and the (partially denoised) action are input tokens to the transformer. Additional conditions can be integrated via self-and cross-attention, e.g., (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>; <xref ref-type="bibr" rid="B103">Mishra and Chen, 2024</xref>). The exact architecture of the transformer varies across methods. The more commonly used model is a multi-head cross-attention transformer as the denoising network, e.g., (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>; <xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>; <xref ref-type="bibr" rid="B170">Wang et al., 2023b</xref>; <xref ref-type="bibr" rid="B103">Mishra and Chen, 2024</xref>). Others (<xref ref-type="bibr" rid="B6">Bharadhwaj et al., 2024b</xref>; <xref ref-type="bibr" rid="B104">Mishra et al., 2023</xref>) use architectures based on the method Diffusion Transformers (<xref ref-type="bibr" rid="B117">Peebles and Xie, 2023</xref>), which is the first method combining DMs with transformer architectures. There are also less commonly used architectures, such as using the output tokens of the transformer as input to an MLP, which predicts the noise (<xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>).</p>
<p>For completeness, we provide a list of works, using transformer architectures: (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>; <xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>; <xref ref-type="bibr" rid="B142">Scheikl et al., 2024</xref>; <xref ref-type="bibr" rid="B170">Wang et al., 2023b</xref>; <xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B30">Feng et al., 2024</xref>; <xref ref-type="bibr" rid="B6">Bharadhwaj et al., 2024b</xref>; <xref ref-type="bibr" rid="B104">Mishra et al., 2023</xref>; <xref ref-type="bibr" rid="B86">Liu et al., 2023b</xref>; <xref ref-type="bibr" rid="B181">Xu et al., 2024</xref>; <xref ref-type="bibr" rid="B103">Mishra and Chen, 2024</xref>; <xref ref-type="bibr" rid="B87">Liu et al., 2023c</xref>; <xref ref-type="bibr" rid="B164">Vosylius et al., 2024</xref>; <xref ref-type="bibr" rid="B130">Reuss et al., 2023</xref>; <xref ref-type="bibr" rid="B52">Iioka et al., 2023</xref>; <xref ref-type="bibr" rid="B50">Huang T. et al., 2025</xref>).</p>
</sec>
<sec id="s3-1-3">
<label>3.1.3</label>
<title>Multi-Layer Perceptrons</title>
<p>Predominantly used for applications in RL, MLPs are employed as denoising networks, e.g., (<xref ref-type="bibr" rid="B156">Suh et al., 2023</xref>; <xref ref-type="bibr" rid="B25">Ding and Jin, 2023</xref>; <xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>), which take concatenated input features, such as observations, actions, and denoising time steps, to predict the noise. Although the architectures vary, it is common to use a relatively small number of hidden layers (2&#x2013;4) (<xref ref-type="bibr" rid="B170">Wang et al., 2023b</xref>; <xref ref-type="bibr" rid="B58">Kang et al., 2023</xref>; <xref ref-type="bibr" rid="B156">Suh et al., 2023</xref>; <xref ref-type="bibr" rid="B99">Mendez-Mendez et al., 2023</xref>), using e.g., Mish activation (<xref ref-type="bibr" rid="B105">Misra, 2019</xref>), following the first method (<xref ref-type="bibr" rid="B169">Wang et al., 2023a</xref>), integrating DMs with Q-learning. It is important to note that most of these methods do not use visual input. An exception from this is <xref ref-type="bibr" rid="B116">Pearce et al. (2022)</xref>, which also evaluates using high-resolution image inputs with an MLP-based DM. However, for this, a CNN-based image encoder is first applied to the raw image observation, before the encoding is fed to the DM.</p>
</sec>
<sec id="s3-1-4">
<label>3.1.4</label>
<title>Comparison</title>
<p>An ongoing debate exists concerning the relative merits of different architectural choices, with each architecture exhibiting distinct advantages and disadvantages. <xref ref-type="bibr" rid="B23">Chi et al. (2023)</xref> implemented both a U-Net-based and a transformer-based denoising network with the application of trajectory planning. They observed that the CNN-based model exhibits lower sensitivity to hyperparameters than transformers. Moreover, they report that when using positional control, the U-net results in a slightly higher success rate for some complex visual tasks, such as transport, tool hand, and push-t. On the other hand, U-nets may induce an over-smoothing effect, thereby resulting in diminished performance for high-frequency trajectories and consequently affecting velocity control. Thus, in these cases, transformers will likely lead to more precise predictions. Furthermore, transformer-based architectures have demonstrated proficiency in capturing long-range dependencies and exhibit notable robustness when handling high-dimensional data, surpassing the abilities of CNNs, which is particularly significant for tasks involving long horizons and high-level decision-making (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>; <xref ref-type="bibr" rid="B27">Dosovitskiy et al., 2021</xref>).</p>
<p>While MLPs typically exhibit inferior performance, especially when confronted with complex problems and high-dimensional input data, such as images, they often demonstrate superior computational efficiency, which facilitates higher-rate sampling and usually requires fewer computational resources. Due to their training stability, they are a commonly used architecture in RL. In contrast, U-Nets, and especially transformers, are characterized by substantial resource consumption and prolonged inference times, which may hinder their application in real-time robotics (<xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>).</p>
<p>In summary, transformers are the most powerful architecture for handling high-dimensional input and output spaces, followed by CNNs, while MLPs have the highest computational efficiency. For processing visual data, such as raw images, an important task in robotic manipulation, a CNN or a Transformer architecture should be chosen. Also, while MLPs are most computationally efficient, real-time control is possible with the other two architectures, integrating, for example, receding horizon control (<xref ref-type="bibr" rid="B96">Mattingley et al., 2011</xref>) in combination with a more efficient sampling process, like DDIM.</p>
</sec>
</sec>
<sec id="s3-2">
<label>3.2</label>
<title>Number of sampling steps</title>
<p>In addition to the network architecture, a crucial decision is the choice of the number of training and sampling iterations. As described in <xref ref-type="sec" rid="s2-2">Section 2.2</xref>, each sample must undergo iterative denoising over several steps, which can be notably time-consuming, especially in the context of employing larger denoising networks with longer inference durations, such as transformers. Within the framework of DDPM, the number of noise levels during training is equal to the number of denoising iterations at the time of inference. This hinders its use in many robotic manipulation scenarios, especially those necessitating real-time predictions. Consequently, numerous methodologies employ DDIM, where the number of sampling iterations during inference can be significantly reduced compared to the number of noise levels used during training. Common choices of noise levels are 50&#x2013;100 during training, but only a subset of five to ten steps during inference (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>; <xref ref-type="bibr" rid="B92">Ma X. et al., 2024</xref>; <xref ref-type="bibr" rid="B50">Huang T. et al., 2025</xref>; <xref ref-type="bibr" rid="B142">Scheikl et al., 2024</xref>). Only a few works used less sampling (3&#x2013;4) (<xref ref-type="bibr" rid="B164">Vosylius et al., 2024</xref>; <xref ref-type="bibr" rid="B130">Reuss et al., 2023</xref>) or more (20&#x2013;30) (<xref ref-type="bibr" rid="B103">Mishra and Chen, 2024</xref>; <xref ref-type="bibr" rid="B167">Wang L. et al., 2024</xref>) sampling steps. <xref ref-type="bibr" rid="B69">Ko et al. (2024)</xref> documented a slight decline in performance when the number of sampling steps is reduced to <inline-formula id="inf70">
<mml:math id="m77">
<mml:mrow>
<mml:mn>10</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> with DDIM (<xref ref-type="bibr" rid="B69">Ko et al., 2024</xref>). Therefore, it is imperative to consider an appropriate trade-off between sample quality and inference time, tailored to the specific task requirements. Still, only a few evaluations exist that compare DDPM-based, DDIM-based, or other samplers for robotic manipulation, and further investigation is required.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Applications</title>
<p>In this section, we explore the most dominant applications of DMs in robotic manipulation: trajectory generation for robotic manipulation, robotic grasping, and visual data augmentation for vision-based robotics manipulations.</p>
<sec id="s4-1">
<label>4.1</label>
<title>Trajectory generation</title>
<p>Trajectory planning in robotic manipulation is vital for enabling robots to move from one point to another smoothly, safely, and efficiently while adhering to physical constraints, like speed and acceleration limits, as well as ensuring collision avoidance. Classical planning methods, like interpolation-based and sampling-based approaches, can have difficulty handling complex tasks or ensuring smooth paths. For instance, Rapidly Exploring Random Trees (<xref ref-type="bibr" rid="B95">Martinez et al., 2023</xref>) might generate trajectories with sudden changes because of the discretization process. As already discussed in the introduction, although popular data-driven approaches, such as GMMs and EBMs, theoretically pertain to the ability to model multi-model data distributions, in reality, they show suboptimal behavior, such as biasing modes or lack of temporal consistency (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>). In addition, GMMs can struggle with high-dimensional input spaces (<xref ref-type="bibr" rid="B45">Ho et al., 2020</xref>). Increasing the number of components and covariances also increases the models&#x2019; ability to model more complex distributions and capture complex and intricate movement patterns. However, this can negatively impact the smoothness of the generated trajectories, making GMMs highly sensitive to their hyperparameters. In contrast, denoising DMs have demonstrated exceptional performance in processing and generating high-dimensional data. Furthermore, the distributions generated by denoising DMs are inherently smooth (<xref ref-type="bibr" rid="B45">Ho et al., 2020</xref>; <xref ref-type="bibr" rid="B151">Sohl-Dickstein et al., 2015</xref>; <xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>). This makes DMs well-suited for complex, high-dimensional scenarios where flexibility and adaptability are required. While most methodologies that apply probabilistic DMs to robotic manipulation focus on imitation learning, they have also been adapted to their application in RL, e.g., (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>; <xref ref-type="bibr" rid="B169">Wang et al., 2023a</xref>).</p>
<p>In the following sections, the methodologies of DMs for trajectory generation will be further discussed and categorized. We will first explain their applications in imitation learning, followed by a discussion on their use in reinforcement learning. For an overview of the method architectures in imitation learning, see <xref ref-type="table" rid="T2">Table 2</xref>, and for reinforcement learning, see <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<sec id="s4-1-1">
<label>4.1.1</label>
<title>Imitation learning</title>
<p>In imitation learning (<xref ref-type="bibr" rid="B188">Zare et al., 2024</xref>), robots attempt to learn a specified task by observing multiple expert demonstrations. This paradigm, commonly known as Learning from Demonstrations (LfD), involves the robot observing expert examples and attempting to replicate the demonstrated behaviors. In this domain, the robot is expected to generalize beyond the specific demonstrations, which allows the robot to adapt to variations in tasks or changes in configuration spaces. This may include diverse observation perspectives, altered environmental conditions, or even new tasks that share structural similarities with those previously demonstrated. Thus, the robot must learn a representation of the task that allows flexibility and skill acquisition beyond the specific scenarios it was trained on. Recent advancements in applying DMs to learn visuomotor policies (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>) enable the generation of smooth action trajectories by modeling the task as a generative process conditioned on sensory observations. Diffusion-based models, initially popularized for high-dimensional data generation such as images and natural languages, have demonstrated significant potential in robotics by effectively learning complex action distributions and generating multi-modal behaviors conditioned on task-specific inputs. For instance, combining with recent progress in multiview transformers (<xref ref-type="bibr" rid="B37">Gervet et al., 2023</xref>; <xref ref-type="bibr" rid="B40">Goyal et al., 2023</xref>) that leverage the foundation model features (<xref ref-type="bibr" rid="B125">Radford et al., 2021</xref>; <xref ref-type="bibr" rid="B112">Oquab et al., 2023</xref>), 3D diffuser actor (<xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>) integrates multi-modal representations to generate the end-effector trajectories. As another example, GNFactor (<xref ref-type="bibr" rid="B189">Ze et al., 2023</xref>) renders multiview features from Stable Diffusion (<xref ref-type="bibr" rid="B133">Rombach et al., 2022b</xref>) to enhance 3d volumetric feature learning. Very similar to diffusion, recently (<xref ref-type="bibr" rid="B137">Rouxel et al., 2024</xref>) flow-matching-based policies have emerged for trajectory generation, generally leading to a more stable training process with fewer hyperparameters, as already mentioned in <xref ref-type="sec" rid="s2-2-1">Section 2.2.1</xref>. <xref ref-type="bibr" rid="B108">Nguyen et al. (2025)</xref> additionally includes second-order dynamics into the flow-matching objective, learning fields on acceleration and jerk to ensure smoothness of the generated trajectories.</p>
<p>In terms of the type of robotic embodiment, most works use parallel grippers or simpler end-effectors. However, few methods perform dexterous manipulation using DMs (<xref ref-type="bibr" rid="B148">Si et al., 2024</xref>; <xref ref-type="bibr" rid="B91">Ma C. et al., 2024</xref>; <xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B19">Chen K. et al., 2024</xref>; <xref ref-type="bibr" rid="B166">Wang C. et al., 2024</xref>; <xref ref-type="bibr" rid="B34">Freiberg et al., 2025</xref>; <xref ref-type="bibr" rid="B172">Welte and Rayyes, 2025</xref>), to facilitate their stability and robustness, also in this high-dimensional setting.</p>
<p>In the following sections, we will first repeat the process of sampling actions for trajectory planning with DMs and discuss common pose representations. Then we shortly address different visual data modalities, in particular 2D vs. 3D visual observations. Afterwards, we look at methods formulating trajectory planning as image generation, before looking at applications in hierarchical, multi-task, and constrained planning, also looking at multi-task planning with vision language action models (VLAs). A visualization of the taxonomy is provided in <xref ref-type="table" rid="T1">Table 1</xref>. More details on the individual method architectures are provided in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Taxonomy of imitation learning approaches for trajectory generation with diffusion models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Perspective</th>
<th align="left">Category</th>
<th align="left">Subcategory</th>
<th align="left">References</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="5" align="left">Methodological</td>
<td rowspan="3" align="left">Actions and pose representations</td>
<td align="center">Task Space &#xa7;4.1.1.1</td>
<td align="center">
<xref ref-type="bibr" rid="B23">Chi et al. (2023)</xref>, <xref ref-type="bibr" rid="B116">Pearce et al. (2022)</xref>, <xref ref-type="bibr" rid="B190">Ze et al. (2024)</xref>, <xref ref-type="bibr" rid="B43">Ha et al. (2023)</xref>, <xref ref-type="bibr" rid="B64">Ke et al. (2024)</xref>, <xref ref-type="bibr" rid="B180">Xu et al. (2023)</xref>, <xref ref-type="bibr" rid="B76">Li et al. (2024c)</xref>, <xref ref-type="bibr" rid="B148">Si et al. (2024)</xref>, <xref ref-type="bibr" rid="B142">Scheikl et al. (2024)</xref>, <xref ref-type="bibr" rid="B179">Xian et al. (2023)</xref>, <xref ref-type="bibr" rid="B87">Liu et al. (2023c)</xref>
</td>
</tr>
<tr>
<td align="center">Joint Space &#xa7;4.1.1.1</td>
<td align="center">
<xref ref-type="bibr" rid="B15">Carvalho et al. (2023)</xref>, <xref ref-type="bibr" rid="B140">Saha et al. (2024)</xref>, <xref ref-type="bibr" rid="B162">Urain et al. (2023)</xref>, <xref ref-type="bibr" rid="B92">Ma et al. (2024b)</xref>
</td>
</tr>
<tr>
<td align="center">Image Space &#xa7;4.1.1.3</td>
<td align="center">
<xref ref-type="bibr" rid="B69">Ko et al. (2024)</xref>, <xref ref-type="bibr" rid="B182">Yang et al. (2024)</xref>, <xref ref-type="bibr" rid="B207">Zhou et al. (2024b)</xref>, <xref ref-type="bibr" rid="B164">Vosylius et al. (2024)</xref>, <xref ref-type="bibr" rid="B28">Du et al. (2023)</xref>, <xref ref-type="bibr" rid="B80">Liang et al. (2024)</xref>
</td>
</tr>
<tr>
<td rowspan="2" align="left">Visual data modality &#xa7;4.1.1.2</td>
<td align="center">2D</td>
<td align="center">e.g. <xref ref-type="bibr" rid="B23">Chi et al. (2023)</xref>, <xref ref-type="bibr" rid="B80">Liang et al. (2024)</xref>, <xref ref-type="bibr" rid="B142">Scheikl et al. (2024)</xref>, <xref ref-type="bibr" rid="B148">Si et al. (2024)</xref>
</td>
</tr>
<tr>
<td align="center">3D</td>
<td align="center">
<xref ref-type="bibr" rid="B72">Li et al. (2025)</xref>, <xref ref-type="bibr" rid="B87">Liu et al. (2023c)</xref>, <xref ref-type="bibr" rid="B166">Wang et al. (2024a)</xref>, <xref ref-type="bibr" rid="B190">Ze et al. (2024)</xref>, <xref ref-type="bibr" rid="B179">Xian et al. (2023)</xref>, <xref ref-type="bibr" rid="B64">Ke et al. (2024)</xref>
</td>
</tr>
<tr>
<td rowspan="5" align="left">Functional</td>
<td rowspan="3" align="left">Long-Horizon and Multi-Task Learning</td>
<td align="center">Hierarchical Planning &#xa7;4.1.1.4</td>
<td align="center">
<xref ref-type="bibr" rid="B192">Zhang et al. (2024a)</xref>, <xref ref-type="bibr" rid="B92">Ma et al. (2024b)</xref>, <xref ref-type="bibr" rid="B179">Xian et al. (2023)</xref>, <xref ref-type="bibr" rid="B43">Ha et al. (2023)</xref>, <xref ref-type="bibr" rid="B51">Huang et al. (2024b)</xref>, <xref ref-type="bibr" rid="B28">Du et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="center">Skill Learning &#xa7;4.1.1.4</td>
<td align="center">
<xref ref-type="bibr" rid="B104">Mishra et al. (2023)</xref>, <xref ref-type="bibr" rid="B67">Kim et al. (2024c)</xref>, <xref ref-type="bibr" rid="B180">Xu et al. (2023)</xref>, <xref ref-type="bibr" rid="B80">Liang et al. (2024)</xref>
</td>
</tr>
<tr>
<td align="center">Vision Language Action Models &#xa7;4.1.1.5</td>
<td align="center">
<xref ref-type="bibr" rid="B113">Pan et al. (2024a)</xref>, <xref ref-type="bibr" rid="B144">Shentu et al. (2024)</xref>, <xref ref-type="bibr" rid="B158">Team et al. (2024)</xref>, <xref ref-type="bibr" rid="B174">Wen et al. (2025)</xref>, <xref ref-type="bibr" rid="B84">Liu et al. (2024)</xref>, <xref ref-type="bibr" rid="B74">Li et al. (2024b)</xref>, <xref ref-type="bibr" rid="B7">Black et al. (2024a)</xref>
</td>
</tr>
<tr>
<td rowspan="2" align="left">Constrained Planning &#xa7;4.1.1.6</td>
<td align="center">Classifier guidance</td>
<td align="center">
<xref ref-type="bibr" rid="B104">Mishra et al. (2023)</xref>, <xref ref-type="bibr" rid="B79">Liang et al. (2023)</xref>, <xref ref-type="bibr" rid="B55">Janner et al. (2022)</xref>, <xref ref-type="bibr" rid="B15">Carvalho et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="center">Classifier-free guidance</td>
<td align="center">
<xref ref-type="bibr" rid="B45">Ho et al. (2021)</xref>, <xref ref-type="bibr" rid="B140">Saha et al. (2024)</xref>, <xref ref-type="bibr" rid="B72">Li et al. (2025)</xref>, <xref ref-type="bibr" rid="B121">Power et al. (2023)</xref>, <xref ref-type="bibr" rid="B129">Reuss et al. (2024a)</xref>, <xref ref-type="bibr" rid="B130">Reuss et al. (2023)</xref>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>Technical details of trajectory diffusion using imitation learning. The references for the encoders are provided in <xref ref-type="sec" rid="s12">Supplementary Appendix Table 1</xref>. In the following, the symbols and abbreviations are explained: H: Whether the method is hierarchical (&#x2713;) or not (&#x2717;). PCs: Point Clouds, Lan: Language, GTS: Ground Truth State, and whether the visual input modality is from single view or (<sup>SV</sup>) multi-view (<sup>MV</sup>). U-Net: temporal U-Net (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>), FiLM: Convolutional Neural Networks with Feature-wise Linear Modulation (<xref ref-type="bibr" rid="B118">Perez et al., 2018</xref>), DiT: Diffusion Transformer, RHC: sub-trajectories with receding horizon control, CT: complete trajectory in task space, J: complete trajectory in joint space. A&#x201c;/&#x201d; indicates that the information is not provided by the cited paper, while a &#x201c;-&#x201d; indicates that no specialized encoder is required as ground truth state information is used.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Reference</th>
<th align="left">Input</th>
<th align="left">Output</th>
<th align="left">Encoder</th>
<th align="left">Diffuser</th>
<th align="center">H</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B23">Chi et al. (2023)</xref>
</td>
<td align="left">
<inline-formula id="inf71">
<mml:math id="m78">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>RGB</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">RHC</td>
<td align="left">ResNet</td>
<td align="left">FiLM</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B179">Xian et al. (2023)</xref>
</td>
<td align="left">RGB-D <sup>MV</sup>, Lan</td>
<td align="left">CT</td>
<td align="left">CLIP</td>
<td align="left">DiT &#x26; MLP</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B130">Reuss et al. (2023)</xref>
</td>
<td align="left">GTS/RGB <sup>SV</sup>
</td>
<td align="left">CT</td>
<td align="left">ResNet</td>
<td align="left">DiT</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B20">Chen et al. (2023a)</xref>
</td>
<td align="left">RGB <sup>SV</sup>, Lan</td>
<td align="left">RHC</td>
<td align="left">ResNet</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B208">Zhou et al. (2023)</xref>
</td>
<td align="left">RGB <sup>MV</sup>
</td>
<td align="left">RHC</td>
<td align="left">CLIP</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B116">Pearce et al. (2022)</xref>
</td>
<td align="left">RGB <sup>SV</sup>
</td>
<td align="left">RHC</td>
<td align="left">CNN/ResNet</td>
<td align="left">MLP/DiT</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B99">Mendez-Mendez et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">MLPs</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B190">Ze et al. (2024)</xref>
</td>
<td align="left">PCs <sup>SV</sup>
</td>
<td align="left">RHC</td>
<td align="left">MLP</td>
<td align="left">FiLM</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B64">Ke et al. (2024)</xref>
</td>
<td align="left">RGB-<inline-formula id="inf72">
<mml:math id="m79">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>D</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>SV/MV</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, Lan</td>
<td align="left">CT</td>
<td align="left">CLIP</td>
<td align="left">DiT</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B121">Power et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">MLP</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B92">Ma et al. (2024b)</xref>
</td>
<td align="left">RGB-D <sup>SV</sup>, Lan</td>
<td align="left">J</td>
<td align="left">PointNet&#x2b;&#x2b;, MLP</td>
<td align="left">U-Net</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B164">Vosylius et al. (2024)</xref>
</td>
<td align="left">RGB <sup>MV</sup>
</td>
<td align="left">RHC</td>
<td align="left">Transformer</td>
<td align="left">DiT</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B192">Zhang et al. (2024a)</xref>
</td>
<td align="left">
<inline-formula id="inf73">
<mml:math id="m80">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>RGB</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>SV</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, Lan</td>
<td align="left">RHC</td>
<td align="left">HULC, T5</td>
<td align="left">U-Net</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B129">Reuss et al. (2024a)</xref>
</td>
<td align="left">RGB <sup>MV</sup>, Lan</td>
<td align="left">RHC</td>
<td align="left">ResNet, CLIP</td>
<td align="left">DiT</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B142">Scheikl et al. (2024)</xref>
</td>
<td align="left">RGB <sup>SV</sup>/GTS</td>
<td align="left">RHC</td>
<td align="left">ResNet</td>
<td align="left">DiT</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B19">Chen et al. (2024a)</xref>
</td>
<td align="left">GTS/PCs/<inline-formula id="inf74">
<mml:math id="m81">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>SV</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">RHC</td>
<td align="left">&#x2014;</td>
<td align="left">&#x2014;</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B206">Zhou et al. (2024a)</xref>
</td>
<td align="left">GTS/<inline-formula id="inf75">
<mml:math id="m82">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mi mathvariant="normal">G</mml:mi>
<mml:mi mathvariant="normal">B</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mtext>SV</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">RHC</td>
<td align="left">ResNet</td>
<td align="left">DiT</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B72">Li et al. (2025)</xref>
</td>
<td align="left">
<inline-formula id="inf76">
<mml:math id="m83">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>PCs</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>SV</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, Lan</td>
<td align="left">RHC</td>
<td align="left">SAM, XMem</td>
<td align="left">FiLM</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B76">Li et al. (2024c)</xref>
</td>
<td align="left">
<inline-formula id="inf77">
<mml:math id="m84">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>RGB</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">RHC</td>
<td align="left">ResNet</td>
<td align="left">FiLM</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B148">Si et al. (2024)</xref>
</td>
<td align="left">
<inline-formula id="inf78">
<mml:math id="m85">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>RGB</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>SV</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">RHC</td>
<td align="left">ResNet</td>
<td align="left">FiLM</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B140">Saha et al. (2024)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B6">Bharadhwaj et al. (2024b)</xref>
</td>
<td align="left">
<inline-formula id="inf79">
<mml:math id="m86">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>RGB</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mtext>SV</mml:mtext>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">point tracks</td>
<td align="left">&#x2014;</td>
<td align="left">DiT</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B167">Wang et al. (2024b)</xref>
</td>
<td align="left">RGB, Tactile, PCs, Lan</td>
<td align="left">RHC</td>
<td align="left">ResNet, PointNet, T5</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B131">Reuss et al. (2024b)</xref>
</td>
<td align="left">RGB, Lan</td>
<td align="left">RHC</td>
<td align="left">ResNet, CLIP</td>
<td align="left">DiT</td>
<td align="center">&#x2717;</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s4-1-1-1">
<label>4.1.1.1</label>
<title>Actions and pose representation</title>
<p>As briefly discussed in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>, the entire trajectory can be generated as a single sample, multiple subsequences can be sampled using receding horizon control, or the trajectory can be generated by sampling individual steps. Only in a few methods (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>; <xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>) the whole trajectory is predicted at once. Although this enables a more efficient prediction, as the denoising has to be performed only once, it prohibits adapting to changes in the environment, requiring better foresight and making it unsuitable for more complex task settings with dynamic or open environments. On the other hand, sampling of individual steps increases the compounding error effect and can negatively affect temporal correlation. Instead of predicting micro-actions, some use DMs to predict waypoints (<xref ref-type="bibr" rid="B145">Shi et al., 2023</xref>). This can decrease the compounding error, by reducing the temporal horizon. However, it relies on preprocessing or task settings that ensure that the space in between waypoints is not occluded. Thus, typically, DMs generate trajectories consisting of sequences of micro-actions represented as end-effector positions, generally encompassing translation and rotation depending on end-effector actuation (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>; <xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B180">Xu et al., 2023</xref>; <xref ref-type="bibr" rid="B76">Li X. et al., 2024</xref>; <xref ref-type="bibr" rid="B148">Si et al., 2024</xref>; <xref ref-type="bibr" rid="B142">Scheikl et al., 2024</xref>; <xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>; <xref ref-type="bibr" rid="B43">Ha et al., 2023</xref>). Once the trajectory is sampled, the proximity of the predicted positions enables computing the motion between the positions with simple positional controllers without the need for complex trajectory planning techniques. The control scheme is visualized in detail in <xref ref-type="fig" rid="F2">Figure 2</xref>. Although more commonly applied in grasp prediction, here the pose is sometimes also represented in special Euclidean group <inline-formula id="inf82">
<mml:math id="m89">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B179">Xian et al., 2023</xref>; <xref ref-type="bibr" rid="B87">Liu et al., 2023c</xref>; <xref ref-type="bibr" rid="B138">Ryu et al., 2024</xref>). Explained in more detail in <xref ref-type="sec" rid="s4-2">Section 4.2</xref>, the group structure of the <inline-formula id="inf83">
<mml:math id="m90">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> Lie group enables continuous interpolation and transformations between multiple object poses. As <xref ref-type="bibr" rid="B87">Liu et al. (2023c)</xref>, <xref ref-type="bibr" rid="B138">Ryu et al. (2024)</xref> performs complex tasks involving trajectory planning and grasping for aligning multiple objects, these properties are important to ensure physically and geometrically grounded actions. However, as the prediction of <inline-formula id="inf84">
<mml:math id="m91">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> poses with DMs requires a more complex model structure and training in imitation learning, it is more usual to use representations, such as Euler angles or quaternions, in trajectory planning. Not only diffusion, but also flow matching has been adapted to use representations in <inline-formula id="inf85">
<mml:math id="m92">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> or Riemannian manifolds in general (<xref ref-type="bibr" rid="B10">Braun et al., 2024</xref>).</p>
<p>Although not common, sometimes actions are predicted directly in joint space (<xref ref-type="bibr" rid="B15">Carvalho et al., 2023</xref>; <xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>; <xref ref-type="bibr" rid="B140">Saha et al., 2024</xref>; <xref ref-type="bibr" rid="B92">Ma X. et al., 2024</xref>), allowing for direct control of joint motions, which, e.g., reduces singularities.</p>
</sec>
<sec id="s4-1-1-2">
<label>4.1.1.2</label>
<title>Visual data modalities</title>
<p>As already discussed in <xref ref-type="sec" rid="s2-3">Section 2.3</xref> to ground the robots actions in the physical world, they are dependent on sensory input. Here, in the majority of methods visual observations are used. In the original work (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>), combining visual robotic manipulation with DMs for trajectory planning, the DM is conditioned on RGB-image observations. Many methods, e.g., (<xref ref-type="bibr" rid="B148">Si et al., 2024</xref>; <xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>; <xref ref-type="bibr" rid="B76">Li X. et al., 2024</xref>), adopt using RGB inputs, also developing more intricate encoding schemes (<xref ref-type="bibr" rid="B123">Qi et al., 2025</xref>).</p>
<p>However, 2D visual scene representations may not provide sufficient geometrical information for intricate robotic tasks, especially in scenes containing occlusions. Thus, multiple later methods used 3D scene representations instead. Here, DMs are either directly conditioned on the point cloud (<xref ref-type="bibr" rid="B72">Li et al., 2025</xref>; <xref ref-type="bibr" rid="B87">Liu et al., 2023c</xref>; <xref ref-type="bibr" rid="B166">Wang C. et al., 2024</xref>) or point cloud feature embeddings (<xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B179">Xian et al., 2023</xref>; <xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>), from singleview (<xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B72">Li et al., 2025</xref>; <xref ref-type="bibr" rid="B166">Wang C. et al., 2024</xref>), or multiview camera setups (<xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>; <xref ref-type="bibr" rid="B179">Xian et al., 2023</xref>). While multiview camera setups provide more complete scene information, they also require a more involved setup and more hardware resources.</p>
<p>These models outperform methods relying solely on 2D visual information, on more complex tasks, also demonstrating robustness to adversarial lighting conditions.</p>
</sec>
<sec id="s4-1-1-3">
<label>4.1.1.3</label>
<title>Trajectory planning as image generation</title>
<p>Another category formulates trajectory generation directly in image space, leveraging the exceptional generative abilities of DMs in image generation. Here (<xref ref-type="bibr" rid="B69">Ko et al., 2024</xref>; <xref ref-type="bibr" rid="B207">Zhou S. et al., 2024</xref>; <xref ref-type="bibr" rid="B28">Du et al., 2023</xref>), given a single image observation, a sequence of images, or a video, sometimes in combination with a language-task-instruction, the diffusion process is conditioned to predict a sequence of images, depicting the change in robot and object position. This comes with the benefit of internet-wide video training data, which facilitates extensive training, leading to good generalization capabilities. Especially in combination with methods (<xref ref-type="bibr" rid="B6">Bharadhwaj et al., 2024b</xref>) agnostic to the robot embodiment, this highly increases the amount of available training data. Moreover, in robotic manipulation, the model usually has to parse visual observations. Predicting actions in image space circumvents the need for mapping from the image space to a usually much lower-dimensional action space, reducing the required amount of training data (<xref ref-type="bibr" rid="B164">Vosylius et al., 2024</xref>). However, predicting high-dimensional images may also prevent the model from successfully learning important details of trajectories, as the DM is not guided to pay more attention to certain regions of the image, even though usually only a low fraction of pixels contain task-relevant information. Additionally, methods generating complete images must ensure temporal consistency and physical plausibility. Hence, extensive training resources are required. As an example (<xref ref-type="bibr" rid="B207">Zhou S. et al., 2024</xref>), uses 100 V100 GPUs and 70k demonstrations for training. While still operating in image space, some methods do not generate whole image sequences, but instead perform point-tracking (<xref ref-type="bibr" rid="B6">Bharadhwaj et al., 2024b</xref>) or diffuse imprecise action-effects on the end-effector position directly in image space (<xref ref-type="bibr" rid="B164">Vosylius et al., 2024</xref>). This mitigates the problem of generating physically implausible scenes. However, point-tracking still requires extensive amounts of data. <xref ref-type="bibr" rid="B6">Bharadhwaj et al. (2024b)</xref>, e.g., uses 0.4 million video clips for training.</p>
</sec>
<sec id="s4-1-1-4">
<label>4.1.1.4</label>
<title>Long-horizon and multi-task learning</title>
<p>Due to their ability to robustly model multi-model distributions and relatively good generalization capabilities, DMs are well suited to handle long-horizon and multi-skill tasks, where usually long-range dependencies and multiple valid solutions exist, especially for high-level task instructions (<xref ref-type="bibr" rid="B99">Mendez-Mendez et al., 2023</xref>; <xref ref-type="bibr" rid="B80">Liang et al., 2024</xref>). Often, long-horizon tasks are modeled using hierarchical structures and skill learning. Usually, a single skill-conditioned DM or several DMs are learned for the individual skills, while the higher-level skill planning does not use a DM (<xref ref-type="bibr" rid="B104">Mishra et al., 2023</xref>; <xref ref-type="bibr" rid="B67">Kim W. K. et al., 2024</xref>; <xref ref-type="bibr" rid="B180">Xu et al., 2023</xref>; <xref ref-type="bibr" rid="B80">Liang et al., 2024</xref>; <xref ref-type="bibr" rid="B75">Li et al., 2023</xref>). The exact architecture for the higher-level skill planning varies across methods, being, for example, a variational autoencoder (<xref ref-type="bibr" rid="B67">Kim W. K. et al., 2024</xref>) or a regression model (<xref ref-type="bibr" rid="B104">Mishra et al., 2023</xref>). Instead of having a separate skill planner that samples one skill, <xref ref-type="bibr" rid="B167">Wang L. et al. (2024)</xref> develops a sampling scheme that can sample from a combination of DMs trained for different tasks and in different settings.</p>
<p>To forego the skill-enumeration, which brings with it the limitation of a predefined finite number of skills, some works employ a coarse-to-fine hierarchical framework, where higher-level policies are used to predict goal states for lower-level policies (<xref ref-type="bibr" rid="B192">Zhang E. et al., 2024</xref>; <xref ref-type="bibr" rid="B92">Ma X. et al., 2024</xref>; <xref ref-type="bibr" rid="B179">Xian et al., 2023</xref>; <xref ref-type="bibr" rid="B43">Ha et al., 2023</xref>; <xref ref-type="bibr" rid="B51">Huang Z. et al., 2024</xref>; <xref ref-type="bibr" rid="B28">Du et al., 2023</xref>).</p>
<p>The ability of DMs to stably process high-dimensional input spaces enables the integration of multi-modal inputs, which is especially important in multi-skill tasks, to develop versatile and generalizable agents via arbitrary skill-chaining. Methodologies use videos (<xref ref-type="bibr" rid="B180">Xu et al., 2023</xref>), images, and natural language task instructions (<xref ref-type="bibr" rid="B80">Liang et al., 2024</xref>; <xref ref-type="bibr" rid="B167">Wang L. et al., 2024</xref>; <xref ref-type="bibr" rid="B207">Zhou S. et al., 2024</xref>; <xref ref-type="bibr" rid="B131">Reuss et al., 2024b</xref>), or even more diverse modalities, such as tactile information and point clouds (<xref ref-type="bibr" rid="B167">Wang L. et al., 2024</xref>), to prompt skills.</p>
<p>Although these methods are designed to enhance generalizability, achieving adaptability in highly dynamic environments and unfamiliar scenarios may require the integration of continuous and lifelong learning. This is a widely unexplored field in the context of DMs, with only very few works (<xref ref-type="bibr" rid="B49">Huang J. et al., 2024</xref>; <xref ref-type="bibr" rid="B26">Di Palo et al., 2024</xref>) exploring this topic. Moreover, these methods are still limited in their applications. <xref ref-type="bibr" rid="B26">Di Palo et al. (2024)</xref> are utilizing a lifelong buffer to accelerate the training of new policies for new tasks. In contrast, <xref ref-type="bibr" rid="B99">Mendez-Mendez et al. (2023)</xref> continually updates its policy. However, they only conduct training and experiments in simulation. Additionally, their method requires precise feature descriptions of all involved objects and is limited to predefined abstract skills. Moreover, for the continual update, all past data is replayed, which is not only computationally inefficient but also does not prevent catastrophic forgetting.</p>
</sec>
<sec id="s4-1-1-5">
<label>4.1.1.5</label>
<title>Multi-task learning with vision language action models</title>
<p>Another approach to enhance generalizability in multi-task settings is the incorporation of pretrained VLAs. As a specialized class of multimodal language model (MLLM), VLAs combine the perceptual and semantic representation power of the vision language foundation model and the motor execution capabilities of the action generation model, thereby forming a cohesive end-to-end decision-making framework. Being pretrained on internet-scale data, VLAs exhibit great generalization capabilities across diverse and unseen scenarios, thereby enabling robots to execute complex tasks with remarkable adaptability (<xref ref-type="bibr" rid="B31">Firoozi et al., 2025</xref>).</p>
<p>A predominant line of approaches among VLAs employs next-token prediction for auto-regressive action token generation, representing a foundational approach to end-to-end VLA modeling, e.g., (<xref ref-type="bibr" rid="B13">Brohan et al., 2023b</xref>; <xref ref-type="bibr" rid="B12">Brohan et al., 2023a</xref>; <xref ref-type="bibr" rid="B65">Kim M. J. et al., 2024</xref>). However, this approach is hindered by significant limitations, most notably the slow inference speeds inherent to auto-regressive methods (<xref ref-type="bibr" rid="B12">Brohan et al., 2023a</xref>; <xref ref-type="bibr" rid="B174">Wen et al., 2025</xref>; <xref ref-type="bibr" rid="B119">Pertsch et al., 2025</xref>). This poses a critical bottleneck for real-time robotic systems, where low-latency decision-making is essential. Furthermore, the discretizations of motion tokens, which reformulates action generation as a classification task, introduces quantization errors that lead to a decrease in control precision, thus reducing the overall performance and reliability (<xref ref-type="bibr" rid="B201">Zhang et al., 2024g</xref>; <xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>; <xref ref-type="bibr" rid="B197">Zhang S. et al., 2024</xref>).</p>
<p>To address these limitations one line of research within VLAs focuses on predicting future states and synthesizing executable actions by leveraging inverse kinematics principles derived from these predictions, e.g., (<xref ref-type="bibr" rid="B18">Cheang et al., 2024</xref>; <xref ref-type="bibr" rid="B204">Zhen et al., 2024</xref>; <xref ref-type="bibr" rid="B195">Zhang et al., 2024c</xref>). While this approach addresses some of the limitations associated with token discretization, multimodal states often correspond to multiple valid actions, and the attempt to model these states through techniques such as arithmetic averaging can result in infeasible or suboptimal action outputs.</p>
<p>Thus, showing strong capabilities and stability in modeling multi-modal distributions, DMs have emerged as a promising solution. Leveraging their strong generalization capabilities, a VLA is used to predict coarse action, while a DM-based policy refines the action, to increase precision and adaptability to different robot embodiments, e.g. (<xref ref-type="bibr" rid="B113">Pan C. et al., 2024</xref>; <xref ref-type="bibr" rid="B144">Shentu et al., 2024</xref>; <xref ref-type="bibr" rid="B158">Team et al., 2024</xref>). For instance, TinyVLA (<xref ref-type="bibr" rid="B174">Wen et al., 2025</xref>) incorporates a diffusion-based head module on top of a pretrained VLA to directly generate robotic actions. More specifically, DP (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>) is connected to the multimodal model backbone via two linear projections and a LayerNorm. The multimodal model backbone jointly encodes the current observations and language instruction, generating a multimodal embedding that conditions and guides the denoising process. Furthermore, in order to better fill the gap between logical reasoning and actionable robot policies, a reasoning injection module is proposed, which reuses reasoning outputs (<xref ref-type="bibr" rid="B173">Wen et al., 2024</xref>). Similarly, conditional diffusion decoders have been leveraged to represent continuous multimodal action distributions, enabling the generation of diverse and contextually appropriate action sequences (<xref ref-type="bibr" rid="B158">Team et al., 2024</xref>; <xref ref-type="bibr" rid="B84">Liu et al., 2024</xref>; <xref ref-type="bibr" rid="B74">Li Q. et al., 2024</xref>).</p>
<p>Addressing the disadvantage of long inference times with DMs, in some recent works instead, flow matching is used to generate actions from observations preprocessed by VLMs to solve flexible and dynamic tasks, offering a robust alternative to traditional diffusion mechanisms (<xref ref-type="bibr" rid="B7">Black et al., 2024a</xref>; <xref ref-type="bibr" rid="B193">Zhang and Gienger, 2025</xref>). While <xref ref-type="bibr" rid="B7">Black et al. (2024a)</xref> takes a skill-based approach, where the vision-language model is used to decide on actions, <xref ref-type="bibr" rid="B193">Zhang and Gienger (2025)</xref> uses a vision-language model to generate waypoints. In both approaches, flow matching is used as the expert policy, generating precise trajectories.</p>
<p>VLAs offer access to models trained on huge amounts of data and with strong computational power, leading to strong generalization capabilities. To mitigate some of their shortcomings, such as imprecise actions, specialized policies can be used for refinement. To not restrict the generalizability of the VLA, DMs offer a great possibility, as they can capture complex multi-model distributions and process high-dimensional visual inputs. However, both VLAs and DMs have a relatively slow inference speed. Thus, especially in this combination with VLAs, increasing the sampling efficiency of DMs is important. One example was provided in the previous paragraph. But the topic of higher sampling speed with DMs is also discussed in more detail in <xref ref-type="sec" rid="s2-2-1">Section 2.2.1</xref>.</p>
</sec>
<sec id="s4-1-1-6">
<label>4.1.1.6</label>
<title>Constrained planning</title>
<p>Another line of methods focuses on constrained trajectory learning. A typical goal is obstacle avoidance, object-centric, or goal-oriented trajectory planning, but other constraints can also be included. If the constraints are known prior to training, they can be integrated into the loss function. However, if the goal is to adhere to various and possibly changing constraints during inference, another approach has to be taken. For less complex constraints, such as specific initial or goal states (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>), introduces a conditioning, where, after each denoising time step (<xref ref-type="disp-formula" rid="e7">Equation 7</xref>), the particular state from the trajectory is replaced by the state from the constraint. However, this can lead the trajectory into regions of low likelihood, hence decreasing stability and potentially causing mode collapse. Moreover, this method is not applicable to more complex constraints.</p>
<p>One approach, also addressed by <xref ref-type="bibr" rid="B55">Janner et al. (2022)</xref>, is classifier guidance (<xref ref-type="bibr" rid="B24">Dhariwal and Nichol, 2021</xref>). Here, a separate model is trained to score the trajectory at each denoising step and steer it toward regions that satisfy the constraint. This is integrated into the denoising process by adding the gradient of the predicted score. It should be noted that for sequential data, such as trajectories, classifier guidance can also bias the sampling towards regions of low likelihood (<xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>). Thus, the weight of the guidance factor must be carefully chosen. Moreover, during the start of the denoising process the guidance model must predict the score on a highly uninformative output (close to Gaussian noise) and should have a lower impact. Therefore, it is important to inform the classifier of the denoising time step, train it also on noisy samples, or adjust the weight with which the guidance factor is integrated into the reverse process. Classifier guidance is applied in several methodologies (<xref ref-type="bibr" rid="B104">Mishra et al., 2023</xref>; <xref ref-type="bibr" rid="B79">Liang et al., 2023</xref>; <xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>; <xref ref-type="bibr" rid="B15">Carvalho et al., 2023</xref>). However, it requires the additional training of a separate model. Furthermore, computing the gradient of the classifier at each sampling step adds additional computational cost. Thus, classifier-free guidance (<xref ref-type="bibr" rid="B46">Ho et al., 2021</xref>; <xref ref-type="bibr" rid="B140">Saha et al., 2024</xref>; <xref ref-type="bibr" rid="B72">Li et al., 2025</xref>; <xref ref-type="bibr" rid="B121">Power et al., 2023</xref>; <xref ref-type="bibr" rid="B129">Reuss et al., 2024a</xref>; <xref ref-type="bibr" rid="B130">Reuss et al., 2023</xref>) has been introduced, where a conditional and an unconditional DM per constraint are trained in parallel. During sampling, a weighted mixture of both DMs is used, allowing for arbitrary combinations of constraints, also not seen together during training. However, it does not generalize to entirely new constraints, as this would necessitate the training of new conditional DMs.</p>
<p>As both classifier and classifier-free guidance only steer the training process, they do not guarantee constraint satisfaction. To guarantee constraint satisfaction in delicate environments, such as surgery (<xref ref-type="bibr" rid="B142">Scheikl et al., 2024</xref>), incorporate movement primitives with DMs to ensure the quality of the trajectory. Recent advances in diffusion models also delve into constraint satisfaction (<xref ref-type="bibr" rid="B134">R&#xf6;mer et al., 2024</xref>), integrating constraint tightening into the reverse diffusion process. While this outperforms previous methods (<xref ref-type="bibr" rid="B121">Power et al., 2023</xref>; <xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>; <xref ref-type="bibr" rid="B16">Carvalho et al., 2024</xref>) in regards to constraint satisfaction, also in multi-constraint settings and constraints not seen during training, the evaluation is done only in simulation on a single experiment setup. Thus, constraint satisfaction with DMs remains an interesting research direction to further explore.</p>
<p>Few methods also perform affordance-based optimization for trajectory planning (<xref ref-type="bibr" rid="B87">Liu et al., 2023c</xref>). However, most work in affordance-based manipulation concentrates on grasp learning, which is discussed in more detail in <xref ref-type="sec" rid="s4-2">Section 4.2</xref>.</p>
</sec>
</sec>
<sec id="s4-1-2">
<label>4.1.2</label>
<title>Offline reinforcement learning</title>
<p>To apply diffusion policies in the context of RL the reward term has to be integrated. Diffuser (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>), one early work adapting diffusion to RL, uses classifier-based guidance, which is based on classifier guidance described in <xref ref-type="sec" rid="s4-1-1-6">Section 4.1.1.6</xref>. Let <inline-formula id="inf86">
<mml:math id="m93">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> be a trajectory with one state-action pair per timestep in a planning horizon <inline-formula id="inf87">
<mml:math id="m94">
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. To incorporate the reward term during sampling, a regression model <inline-formula id="inf88">
<mml:math id="m95">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is trained to predict the return, i.e., the cumulative future reward, over the trajectory <inline-formula id="inf89">
<mml:math id="m96">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> at each denoising time step <inline-formula id="inf90">
<mml:math id="m97">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. This is incorporated into the sampling process by adding the guidance term at each iteration of the reverse diffusion process (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>):<disp-formula id="e8">
<mml:math id="m98">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x2223;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">O</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2248;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
<mml:mi>&#x2207;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">&#x3a3;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>
</p>
<p>Moreover, to ensure that the current state observation <inline-formula id="inf91">
<mml:math id="m99">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is not changed by the reverse diffusion on the trajectory, <inline-formula id="inf92">
<mml:math id="m100">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> is set to the current state observation after each reverse diffusion iteration. In the same way, goal-conditioning or other constraints, which can be accomplished by replacing states from the trajectory with states from the constraint, can be integrated into the method. This, is done in several methodologies (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>; <xref ref-type="bibr" rid="B79">Liang et al., 2023</xref>). However, it has to be done with care, as it can lead to trajectories in regions of low likelihood which may cause instability and mode-collapse (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>; <xref ref-type="bibr" rid="B155">Song Y. et al., 2021</xref>). After the reverse process is completed and <inline-formula id="inf93">
<mml:math id="m101">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> has been predicted, the first action <inline-formula id="inf94">
<mml:math id="m102">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> of the plan is executed. Then, the planning horizon is shifted one step forward, and the next action is sampled.</p>
<p>In Diffuser (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>) and Diffuser-based methods (<xref ref-type="bibr" rid="B156">Suh et al., 2023</xref>; <xref ref-type="bibr" rid="B79">Liang et al., 2023</xref>), the DM is trained independently of the reward signal, similar to methods in imitation learning with DM. Not leveraging the reward signal for training the policy can lead to misalignment of the learned trajectories with optimal trajectories and thus suboptimal behavior of the policy. In contrast, leveraging the reward signal already during training of the policy, can steer the training process, consequently increasing both quality of the trained policy and sample efficiency.</p>
<p>To mitigate these shortcomings, one approach, Decision Diffuser (<xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>), directly conditions the DM on the return of the trajectory using classifier-free guidance. This method outperforms Diffuser on a variety of tasks, such a block-stacking task. However, both methods have not been evaluated on real-world tasks. Directly conditioning on the return, limits generalization capabilities. Different to Q-learning, where the value function is approximated, which generalizes across all future trajectories, here only the return of the current trajectory is considered. Sharing some similarity to on-policy methods, this limits generalization as the policy learns to follow trajectories from the demonstrations with high return values. Thus, this can also be interpreted as guided imitation learning.</p>
<p>A more common method (<xref ref-type="bibr" rid="B169">Wang et al., 2023a</xref>) integrates offline Q-learning with DMs. The loss function from <xref ref-type="disp-formula" rid="e5">Equation 5</xref> is a behavior cloning loss, as the goal is to minimize error with respect to samples taken via the behavior policy. <xref ref-type="bibr" rid="B169">Wang et al. (2023a)</xref> suggests including a critic in the training procedure, which they call Diffusion Q-learning (Diffusion-QL). In Diffusion-QL a Q-function is trained, by minimizing the Bellman-Operator using the double Q-learning trick. The actions for updating the Q-function are sampled from the DM. In turn a policy improvement step <inline-formula id="inf95">
<mml:math id="m103">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="bold-italic">s</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:mi mathvariant="script">D</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold-italic">s</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold-italic">a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is included in the loss for updating the DM (<xref ref-type="bibr" rid="B169">Wang et al., 2023a</xref>):<disp-formula id="e9">
<mml:math id="m104">
<mml:mrow>
<mml:mtable class="aligned">
<mml:mtr>
<mml:mtd columnalign="right">
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>arg</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>where <inline-formula id="inf96">
<mml:math id="m105">
<mml:mrow>
<mml:mi mathvariant="script">L</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the diffusion loss from <xref ref-type="disp-formula" rid="e5">Equation 5</xref> and the parameter <inline-formula id="inf97">
<mml:math id="m106">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> regulates the influence of the critic. Several methods (<xref ref-type="bibr" rid="B1">Ada et al., 2024</xref>; <xref ref-type="bibr" rid="B66">Kim S. et al., 2024</xref>; <xref ref-type="bibr" rid="B163">Venkatraman et al., 2023</xref>; <xref ref-type="bibr" rid="B58">Kang et al., 2023</xref>), build on Diffusion Q-learning. To increase the generalizability to out-of-distribution data, a common problem in offline RL (<xref ref-type="bibr" rid="B71">Levine et al., 2020</xref>), <xref ref-type="bibr" rid="B1">Ada et al. (2024)</xref>, include a state-reconstruction loss, into the training of the DM. An overview of the architectures of methods combining diffusion and reinforcement learning is provided in <xref ref-type="table" rid="T3">Table 3</xref>.</p>
<table-wrap id="T3" position="float">
<label>TABLE 3</label>
<caption>
<p>Technical details of trajectory diffusion using reinforcement learning. The references for the encoders are provided in <xref ref-type="sec" rid="s12">Supplementary Appendix Table 1</xref>. In the following, the symbols and abbreviations are explained: H/S: Whether the method is hierarchical/skill-based (&#x2713;) or not (&#x2717;). Lan: Language, GTS: Ground Truth State, and whether the visual input modality is from single view (<sup>SV</sup>) or multi-view (<sup>MV</sup>). U-Net: temporal U-Net (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>), Eq.: Equivariant FiLM: Convolutional Neural Networks with Feature-wise Linear Modulation (<xref ref-type="bibr" rid="B118">Perez et al., 2018</xref>), DiT: Diffusion Transformer, RHC: sub-trajectories with receding horizon control, Sia &#x3d; single actions. A &#x201c;-&#x201d; indicates that no specialized encoder is required as ground truth state information is used.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Reference</th>
<th align="left">Input</th>
<th align="left">Output</th>
<th align="left">Encoder</th>
<th align="left">Diffuser</th>
<th align="center">H/S</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B55">Janner et al. (2022)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B2">Ajay et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">U-Net</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B169">Wang et al. (2023a)</xref>
</td>
<td align="left">GTS</td>
<td align="left">SiA</td>
<td align="left">-</td>
<td align="left">MLP</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B170">Wang et al. (2023b)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">DiT</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B25">Ding and Jin (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">SiA</td>
<td align="left">-</td>
<td align="left">MLP</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B104">Mishra et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">DiT</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B58">Kang et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">MLP</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B11">Brehmer et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">Eq. U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B156">Suh et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B43">Ha et al. (2023)</xref>
</td>
<td align="left">
<inline-formula id="inf80">
<mml:math id="m87">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>RGB</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, Lan</td>
<td align="left">RHC</td>
<td align="left">ResNet, CLIP</td>
<td align="left">FiLM</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B66">Kim et al. (2024b)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">U-Net</td>
<td align="center">&#x2713;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B79">Liang et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B1">Ada et al. (2024)</xref>
</td>
<td align="left">GTS</td>
<td align="left">SiA</td>
<td align="left">-</td>
<td align="left">MLP</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B128">Ren et al. (2024)</xref>
</td>
<td align="left">RGB/GTS</td>
<td align="left">SiA</td>
<td align="left">ViT/-</td>
<td align="left">U-Net/MLP</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B50">Huang et al. (2025b)</xref>
</td>
<td align="left">
<inline-formula id="inf81">
<mml:math id="m88">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>RGB</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>V</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">SiA</td>
<td align="left">VQ-GAN</td>
<td align="left">VQ-Diffusion <xref ref-type="bibr" rid="B41">Gu et al. (2022)</xref>
</td>
<td align="center">&#x2717;</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B15">Carvalho et al. (2023)</xref>
</td>
<td align="left">GTS</td>
<td align="left">RHC</td>
<td align="left">-</td>
<td align="left">U-Net</td>
<td align="center">&#x2717;</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>One characteristic of methodologies combining RL with DMs is that they are offline methods, with both the policy, i.e., the DM, and the return prediction model/critic being trained offline. This introduces the usual advantages and disadvantages of offline RL (<xref ref-type="bibr" rid="B71">Levine et al., 2020</xref>). The model relies on high-quality existing data, consisting of state-action-reward transitions, and is unable to react to distribution shifts. If not tuned well, this may also lead to overfitting. On the other hand, it has increased sample efficiency and does not require real-time data collections and training, which decreases computational cost and can increase training stability. Compared to imitation learning (<xref ref-type="bibr" rid="B71">Levine et al., 2020</xref>; <xref ref-type="bibr" rid="B120">Pfrommer et al., 2024</xref>; <xref ref-type="bibr" rid="B44">Ho and Ermon, 2016</xref>), offline RL requires data labeled with rewards, the training of a reward function, and is more prone to overfitting to suboptimal behavior. However, confronted with data containing diverse and suboptimal behavior, offline RL has the potential of better generalization compared to imitation learning, as it is well suited to model the entire state-action space. Thus, combining RL with DMs has the potential of modeling highly multi-modal distributions over the whole state-action space, strongly increasing generalizability (<xref ref-type="bibr" rid="B79">Liang et al., 2023</xref>; <xref ref-type="bibr" rid="B128">Ren et al., 2024</xref>). In contrast, if high-quality expert demonstrations are available, imitation learning might lead to better performance and computational efficiency. To overcome some of the shortcoming of imitation learning, such as the covariate shift problem (<xref ref-type="bibr" rid="B136">Ross and Bagnell, 2010</xref>), which make it difficult to handle out of distribution situations, some strategies are devised to finetune behavior cloning policies using RL (<xref ref-type="bibr" rid="B128">Ren et al., 2024</xref>; <xref ref-type="bibr" rid="B50">Huang T. et al., 2025</xref>).</p>
<p>Skill-composition is a common method, to handle long-horizon tasks. To leverage the abilities of RL to learn from suboptimal behaviors multiple methodologies (<xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>; <xref ref-type="bibr" rid="B67">Kim W. K. et al., 2024</xref>; <xref ref-type="bibr" rid="B163">Venkatraman et al., 2023</xref>; <xref ref-type="bibr" rid="B66">Kim S. et al., 2024</xref>) combine skill-learning and RL with DMs.</p>
<p>Only little research (<xref ref-type="bibr" rid="B25">Ding and Jin, 2023</xref>; <xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>) in online and offline-to-online RL with DMs has been conducted, leaving a wide field open for research. Moreover, in the context of skill-learning (<xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>), the DMs, used for the lower-level policies, are trained offline and remain frozen, while the higher-level policy are trained using online RL.</p>
<p>It should be noted that, apart from <xref ref-type="bibr" rid="B128">Ren et al. (2024)</xref>; <xref ref-type="bibr" rid="B50">Huang T. et al. (2025)</xref>, none of the aforementioned methods process visual observations and instead rely on ground-truth environment information, which is only easily available in simulation. Moreover, while all methods have also been tested on robotic manipulation tasks, only a few (<xref ref-type="bibr" rid="B128">Ren et al., 2024</xref>; <xref ref-type="bibr" rid="B50">Huang T. et al., 2025</xref>) have been deliberately engineered for these specific applications. Expanding the scope to encompass all methodologies devised for robotics at large, there is a more substantial body of work that integrates diffusion policies with RL.</p>
</sec>
</sec>
<sec id="s4-2">
<label>4.2</label>
<title>Robotic grasp generation</title>
<p>Grasp learning, as one of the crucial skills for robotic manipulation, has been studied over decades (<xref ref-type="bibr" rid="B107">Newbury et al., 2023</xref>). Starting from hand-crafted feature engineering to statistical approaches (<xref ref-type="bibr" rid="B9">Bohg et al., 2013</xref>), accompanied by the recent progress in deep neural networks that are powered by massive data collection either from real-world (<xref ref-type="bibr" rid="B29">Fang et al., 2020</xref>) or simulated environments (<xref ref-type="bibr" rid="B38">Gilles et al., 2023</xref>; <xref ref-type="bibr" rid="B39">Gilles et al., 2025</xref>; <xref ref-type="bibr" rid="B146">Shi et al., 2024</xref>). The current trend in grasp learning incorporates semantic-level object detection, leveraging open-vocabulary foundation models (<xref ref-type="bibr" rid="B125">Radford et al., 2021</xref>; <xref ref-type="bibr" rid="B85">Liu et al., 2025</xref>), and focuses on object-centric or affordance-based grasp detection in the wild (<xref ref-type="bibr" rid="B124">Qian et al., 2024</xref>; <xref ref-type="bibr" rid="B147">Shi et al., 2025</xref>). To this end, DMs, known for their ability to model complex distributions, allow for the creation of diverse and realistic grasp scenarios by simulating possible interactions with objects in a variety of contexts (<xref ref-type="bibr" rid="B133">Rombach et al., 2022b</xref>). Furthermore, these models contribute to direct grasp generation by optimizing the generation of feasible and efficient grasps (<xref ref-type="bibr" rid="B162">Urain et al., 2023</xref>), particularly in environments where real-time decision-making and adaptability are critical.</p>
<p>Grasp generation with DMs can be categorized into several key approaches: From methodological perspective, one category focuses on explicit diffusion on 6-DoF grasp poses that lie on the <inline-formula id="inf98">
<mml:math id="m107">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> group, directly modeling spatial transformations to generate feasible grasps (<xref ref-type="bibr" rid="B162">Urain et al., 2023</xref>; <xref ref-type="bibr" rid="B153">Song et al., 2024</xref>; <xref ref-type="bibr" rid="B178">Wu et al., 2024b</xref>; <xref ref-type="bibr" rid="B175">Weng et al., 2024</xref>; <xref ref-type="bibr" rid="B150">Singh et al., 2024</xref>; <xref ref-type="bibr" rid="B81">Lim et al., 2024</xref>). Another line of approaches involves implicit grasp diffusion within latent space, enhancing adaptability and versatility (<xref ref-type="bibr" rid="B4">Barad et al., 2024</xref>). A recent trend focuses on language-guided diffusion for task-oriented grasp generation, where natural language inputs shape the generation process (<xref ref-type="bibr" rid="B109">Nguyen N. et al., 2024</xref>; <xref ref-type="bibr" rid="B165">Vuong et al., 2024</xref>; <xref ref-type="bibr" rid="B110">Nguyen T. et al., 2024</xref>; <xref ref-type="bibr" rid="B17">Chang and Sun, 2024</xref>). Other approaches emphasize affordance-driven diffusion, targeting specific functional goals, such as object pose diffusion for rearrangement (<xref ref-type="bibr" rid="B86">Liu et al., 2023b</xref>; <xref ref-type="bibr" rid="B203">Zhao et al., 2025</xref>), affordance-guided object reorientation (<xref ref-type="bibr" rid="B103">Mishra and Chen, 2024</xref>), imitation learning (<xref ref-type="bibr" rid="B176">Wu et al., 2024a</xref>; <xref ref-type="bibr" rid="B91">Ma C. et al., 2024</xref>) or multi-embodiment grasping (<xref ref-type="bibr" rid="B34">Freiberg et al., 2025</xref>). Apart from these categories, hand-object interaction (HOI) specifically prioritizes the synthesis of realistic, functional interactions by modeling the hand&#x2019;s adaptive responses to various object shapes and affordances with dexterity (<xref ref-type="bibr" rid="B184">Ye et al., 2024</xref>; <xref ref-type="bibr" rid="B168">Wang Y.-K. et al., 2024</xref>; <xref ref-type="bibr" rid="B196">Zhang et al., 2024d</xref>; <xref ref-type="bibr" rid="B14">Cao et al., 2024</xref>; <xref ref-type="bibr" rid="B73">Li P. et al., 2024</xref>; <xref ref-type="bibr" rid="B200">Zhang et al., 2025</xref>; <xref ref-type="bibr" rid="B89">Lu et al., 2025</xref>; <xref ref-type="bibr" rid="B194">Zhang et al., 2024b</xref>). In addition to the diffusion on grasp generation or trajectory planning, DM as sim-to-real generator (<xref ref-type="bibr" rid="B78">Li Y. et al., 2024</xref>) or foundational feature extractor (<xref ref-type="bibr" rid="B161">Tsagkas et al., 2024</xref>) such as stable diffusion (<xref ref-type="bibr" rid="B132">Rombach et al., 2022a</xref>) may provide semantic information to enhance downstream grasp generation tasks. <xref ref-type="table" rid="T4">Table 4</xref> summarizes the aforementioned categories. Notably, we include the applications of diffusion in HOI, imitation learning for pre-grasp, and tasks related to image generation in the graph, which will not be further discussed in the rest of this survey due to their relevance to the field of computer vision. While readers are still encouraged to refer to the relevant literature according to our illustration (<xref ref-type="table" rid="T4">Table 4</xref>: HOI Synthesis). More details on the architectures of the individual methods in grasp learning are provided in <xref ref-type="table" rid="T5">Table 5</xref>.</p>
<table-wrap id="T4" position="float">
<label>TABLE 4</label>
<caption>
<p>Taxonomy of grasp generation approaches with diffusion models.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Perspective</th>
<th align="left">Category</th>
<th align="left">Subcategory</th>
<th align="left">References</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td rowspan="4" align="left">Methodological</td>
<td rowspan="2" align="left">Diffusion on <inline-formula id="inf99">
<mml:math id="m108">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> grasp poses</td>
<td align="center">Parallel jaw grasp</td>
<td align="center">
<xref ref-type="bibr" rid="B162">Urain et al. (2023)</xref>, <xref ref-type="bibr" rid="B153">Song et al. (2024)</xref>, <xref ref-type="bibr" rid="B150">Singh et al. (2024)</xref>, <xref ref-type="bibr" rid="B81">Lim et al. (2024)</xref>, <xref ref-type="bibr" rid="B16">Carvalho et al. (2024)</xref>, <xref ref-type="bibr" rid="B138">Ryu et al. (2024)</xref>, <xref ref-type="bibr" rid="B34">Freiberg et al. (2025)</xref>, <xref ref-type="bibr" rid="B47">Huang et al. (2025a)</xref>
</td>
</tr>
<tr>
<td align="center">Dextrous grasp</td>
<td align="center">
<xref ref-type="bibr" rid="B178">Wu et al. (2024b)</xref>, <xref ref-type="bibr" rid="B175">Weng et al. (2024)</xref>, <xref ref-type="bibr" rid="B168">Wang et al. (2024c)</xref>, <xref ref-type="bibr" rid="B34">Freiberg et al. (2025)</xref>, <xref ref-type="bibr" rid="B205">Zhong and Allen-Blanchette (2025)</xref>, <xref ref-type="bibr" rid="B202">Zhang et al. (2024h)</xref>, <xref ref-type="bibr" rid="B177">Wu et al. (2023)</xref>
</td>
</tr>
<tr>
<td align="center">Diffusion in latent space</td>
<td align="center">-</td>
<td align="center">
<xref ref-type="bibr" rid="B4">Barad et al. (2024)</xref>
</td>
</tr>
<tr>
<td align="center">Diffusion as feature encoders and image generators</td>
<td align="center">-</td>
<td align="center">
<xref ref-type="bibr" rid="B78">Li et al. (2024d)</xref>, <xref ref-type="bibr" rid="B161">Tsagkas et al. (2024)</xref>
</td>
</tr>
<tr>
<td rowspan="4" align="left">Functional</td>
<td rowspan="2" align="left">Affordance-driven diffusion</td>
<td align="center">Language-guided grasp diffusion</td>
<td align="center">
<xref ref-type="bibr" rid="B109">Nguyen et al. (2024a)</xref>, <xref ref-type="bibr" rid="B165">Vuong et al. (2024)</xref>, <xref ref-type="bibr" rid="B110">Nguyen et al. (2024b)</xref>, <xref ref-type="bibr" rid="B17">Chang and Sun (2024)</xref>, <xref ref-type="bibr" rid="B200">Zhang et al. (2025)</xref>
</td>
</tr>
<tr>
<td align="center">Pre-grasp manipulation via imitation learning</td>
<td align="center">
<xref ref-type="bibr" rid="B176">Wu et al. (2024a)</xref>, <xref ref-type="bibr" rid="B91">Ma et al. (2024a)</xref>
</td>
</tr>
<tr>
<td align="center">HOI synthesis</td>
<td align="center">-</td>
<td align="center">
<xref ref-type="bibr" rid="B184">Ye et al. (2024)</xref>, <xref ref-type="bibr" rid="B168">Wang et al. (2024c)</xref>, <xref ref-type="bibr" rid="B196">Zhang et al. (2024d)</xref>, <xref ref-type="bibr" rid="B14">Cao et al. (2024)</xref>, <xref ref-type="bibr" rid="B73">Li et al. (2024a)</xref>, <xref ref-type="bibr" rid="B200">Zhang et al. (2025)</xref>, <xref ref-type="bibr" rid="B89">Lu et al. (2025)</xref>
</td>
</tr>
<tr>
<td align="center">Object pose diffusion for reorientation and rearrangement</td>
<td align="center">-</td>
<td align="center">
<xref ref-type="bibr" rid="B86">Liu et al. (2023b)</xref>, <xref ref-type="bibr" rid="B149">Simeonov et al. (2023)</xref>, <xref ref-type="bibr" rid="B103">Mishra and Chen (2024)</xref>, <xref ref-type="bibr" rid="B203">Zhao et al. (2025)</xref>
</td>
</tr>
</tbody>
</table>
</table-wrap>
<table-wrap id="T5" position="float">
<label>TABLE 5</label>
<caption>
<p>Technical details of grasp diffusion methodologies on <inline-formula id="inf100">
<mml:math id="m109">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> grasp synthesis. The references for the encoders are provided in <xref ref-type="sec" rid="s12">Supplementary Appendix Table 1</xref>. The references for the benchmarks are listed in <xref ref-type="sec" rid="s12">Supplementary Appendix Table 2</xref>. In the following, the abbreviations used are explained: SDF: Signed Distance Function, TSDF: Truncated SDF, PCs: Point Clouds, FiLM: Convolutional Neural Network with Feature-wise Linear Modulation (<xref ref-type="bibr" rid="B118">Perez et al., 2018</xref>), DiTs: Diffusion Transformers, Eq.: Equivariant, VN: Vector Neuron.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Reference</th>
<th align="left">Input</th>
<th align="left">Encoder</th>
<th align="left">Diffuser</th>
<th align="left">Benchmark</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">
<xref ref-type="bibr" rid="B162">Urain et al. (2023)</xref>
</td>
<td align="left">SDF</td>
<td align="left">Shape encoder</td>
<td align="left">FiLM</td>
<td align="left">Acronym</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B4">Barad et al. (2024)</xref>
</td>
<td align="left">PCs</td>
<td align="left">PointNet&#x2b;&#x2b;</td>
<td align="left">FiLM</td>
<td align="left">Acronym</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B153">Song et al. (2024)</xref>
</td>
<td align="left">TSDF</td>
<td align="left">OccNet</td>
<td align="left">FiLM</td>
<td align="left">VGN</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B150">Singh et al. (2024)</xref>
</td>
<td align="left">PCs</td>
<td align="left">OccNet</td>
<td align="left">FiLM</td>
<td align="left">
<inline-formula id="inf101">
<mml:math id="m110">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>DA</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B81">Lim et al. (2024)</xref>
</td>
<td align="left">PCs</td>
<td align="left">VN-DGCNN</td>
<td align="left">FiLM</td>
<td align="left">Acronym</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B34">Freiberg et al. (2025)</xref>
</td>
<td align="left">PCs &#x2b; Gripper PCs</td>
<td align="left">Eq. U-Net</td>
<td align="left">Eq. FiLM</td>
<td align="left">Self generated</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B16">Carvalho et al. (2024)</xref>
</td>
<td align="left">PCs</td>
<td align="left">PointNet&#x2b;&#x2b;</td>
<td align="left">DiT</td>
<td align="left">Acronym</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B47">Huang et al. (2025a)</xref>
</td>
<td align="left">PCs &#x2b; Guidance</td>
<td align="left">VN-PointNet</td>
<td align="left">DiTs</td>
<td align="left">OakInk</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B175">Weng et al. (2024)</xref>
</td>
<td align="left">PCs &#x2b; Gripper PCs</td>
<td align="left">BPS</td>
<td align="left">DiTs</td>
<td align="left">DexGraspNet</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B205">Zhong and Allen-Blanchette (2025)</xref>
</td>
<td align="left">PCs &#x2b; Gripper PCs</td>
<td align="left">Eq. Models</td>
<td align="left">Eq. DiTs</td>
<td align="left">MultiDex</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B202">Zhang et al. (2024h)</xref>
</td>
<td align="left">PCs</td>
<td align="left">PointNet&#x2b;&#x2b;</td>
<td align="left">DiTs</td>
<td align="left">MultiDex</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s4-2-1">
<label>4.2.1</label>
<title>Diffusion as <inline-formula id="inf102">
<mml:math id="m111">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> grasp pose generation</title>
<p>Since the standard diffusion process is primarily formulated in Euclidean space, directly extending it to <inline-formula id="inf103">
<mml:math id="m112">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> poses, represented by: <inline-formula id="inf104">
<mml:math id="m113">
<mml:mrow>
<mml:mi mathvariant="bold">H</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mtable class="matrix">
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mi mathvariant="bold">R</mml:mi>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd columnalign="center">
<mml:mn mathvariant="bold">0</mml:mn>
</mml:mtd>
<mml:mtd columnalign="center">
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula> is inherently challenging due to potential numerical instability (to satisfy <inline-formula id="inf105">
<mml:math id="m114">
<mml:mrow>
<mml:mi mathvariant="bold">H</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">H</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="bold">I</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>), since typical Langevin dynamics cannot be applied for non-Euclidean manifolds such as the <inline-formula id="inf106">
<mml:math id="m115">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> Lie group. Here, <inline-formula id="inf107">
<mml:math id="m116">
<mml:mrow>
<mml:mi mathvariant="bold">R</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents the rotation matrix and <inline-formula id="inf108">
<mml:math id="m117">
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> the translation vector. Applying diffusion to <inline-formula id="inf109">
<mml:math id="m118">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> poses requires accounting for the manifold&#x2019;s non-Euclidean nature, where standard Gaussian noise, as used in vanilla diffusion, fails to retain stability over rotations and translations.</p>
<p>To tackle this, <inline-formula id="inf110">
<mml:math id="m119">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>-Diff (<xref ref-type="bibr" rid="B162">Urain et al., 2023</xref>) introduced a smooth cost function to learn the grasp quality via the energy-based model (EBM), where the score matching for EBM is applied on the Lie group to bridge the gap between diffusion processes on the vector space <inline-formula id="inf111">
<mml:math id="m120">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and the <inline-formula id="inf112">
<mml:math id="m121">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In contrast, <xref ref-type="bibr" rid="B153">Song et al. (2024)</xref> condition the 6-Dof grasp poses on the grasp locations <inline-formula id="inf113">
<mml:math id="m122">
<mml:mrow>
<mml:mi mathvariant="bold">t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and corresponding volumetric features for grasp generation in clutter following GIGA framework (<xref ref-type="bibr" rid="B56">Jiang et al., 2021</xref>), without explicit consideration on the <inline-formula id="inf114">
<mml:math id="m123">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> constraint. Moreover, one advantage of the EBM model in <inline-formula id="inf115">
<mml:math id="m124">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>-Diff is the direct grasp quality evaluation and integration into the entire grasp motion planning and optimization. However, training EBM-based models demands extensive sampling and poses significant challenges for generalization. We noticed that flow matching (<xref ref-type="bibr" rid="B82">Lipman et al., 2023</xref>) is employed in recent studies, such as EquiGraspFlow (<xref ref-type="bibr" rid="B81">Lim et al., 2024</xref>) and Grasp Diffusion Network (<xref ref-type="bibr" rid="B16">Carvalho et al., 2024</xref>), which use continuous normalizing flows (CNFs) as ODE solvers to learn angular <inline-formula id="inf116">
<mml:math id="m125">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">O</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and linear <inline-formula id="inf117">
<mml:math id="m126">
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">R</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> velocities for denoising. This preserves the <inline-formula id="inf118">
<mml:math id="m127">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>-equivariance conditioned on the input point cloud given the time schedule. In contrast to <inline-formula id="inf119">
<mml:math id="m128">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>-Diff, which relies on additional supervision in the form of signed distance functions, they achieve competitive performance without requiring this auxiliary module, leading to more efficient training. In general, although CNF-based approaches exhibit promising performance on grasp generation for a single object, more studies on generalizability to highly occluded environments (<xref ref-type="bibr" rid="B34">Freiberg et al., 2025</xref>) and uncertainty quantification (<xref ref-type="bibr" rid="B146">Shi et al., 2024</xref>) are expected in future work.</p>
<p>In contrast to explicit pose diffusion, latent DMs for grasp generation (GraspLDM (<xref ref-type="bibr" rid="B4">Barad et al., 2024</xref>)) explore latent space diffusion with VAEs, which does not explicitly account for the <inline-formula id="inf120">
<mml:math id="m129">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> constraint. They follow VAE-based 6-Dof Graspnet (<xref ref-type="bibr" rid="B106">Mousavian et al., 2019</xref>) to model the distribution of grasp latent features by a denoising diffusion process, which is conditioned on the point cloud and task latent for the grasp generation. This implicit modeling may potentially limit the model&#x2019;s ability to generate physically plausible and geometrically consistent grasp poses.</p>
<p>Furthermore, the <inline-formula id="inf121">
<mml:math id="m130">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> bi-equivariance property is critical for efficient grasp generation (<xref ref-type="bibr" rid="B48">Huang et al., 2023</xref>), as it requires that any transformation applied to the input space correspondingly transforms the output space in a consistent manner. Specifically, this property implies that the generated poses from a <inline-formula id="inf122">
<mml:math id="m131">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>-invariant distribution should maintain the same spatial and geometric relationships under transformations over the time schedule, ensuring that the learned grasp distribution remains invariant across various orientations and positions. For instance, <xref ref-type="bibr" rid="B139">Ryu et al. (2023)</xref> consider bi-equivariance in Lie group representation to construct the equivariance descriptor field (EDF) (<xref ref-type="bibr" rid="B139">Ryu et al., 2023</xref>), taking the transformations of both observation (target) space and initial end-effector frame into account. This principally improves the sample efficiency on pick-and-place tasks via Imitation learning. Upon this, they extend the EDF to bi-equivariant score matching (<xref ref-type="bibr" rid="B138">Ryu et al., 2024</xref>) to be applied in the context of diffusion, which consists of both translational and rotational fields on <inline-formula id="inf123">
<mml:math id="m132">
<mml:mrow>
<mml:mi mathvariant="bold">s</mml:mi>
<mml:mi mathvariant="bold">e</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> Lie algebra. Moreover, <xref ref-type="bibr" rid="B34">Freiberg et al. (2025)</xref> adapts the approach from <xref ref-type="bibr" rid="B138">Ryu et al. (2024)</xref> to generalize to multi-embodiment grasping through an equivariant encoder that captures gripper embeddings. In terms of the theoretical background to equivariant robot learning, we identify a recent survey (<xref ref-type="bibr" rid="B143">Seo et al., 2025</xref>) as a recommendation for interested readers.</p>
</sec>
</sec>
<sec id="s4-3">
<label>4.3</label>
<title>Visual data augmentation</title>
<p>One line of methodologies focuses on employing mostly pretrained DMs for data augmentation in vision-based manipulation tasks. Here, the strong image generation and processing capabilities of diffusion generative models are utilized to augment data sets and scenes. The main goals of the visual data augmentation are scaling up data sets, scene reconstruction, and scene rearrangement.</p>
<sec id="s4-3-1">
<label>4.3.1</label>
<title>Scaling data and scene augmentation</title>
<p>A challenge associated with data-driven approaches in robotics relates to substantial data requirements, which are time-consuming to acquire, particularly for real-world data. In the domain of imitation learning, it is essential to accumulate an adequate number of expert demonstrations that accurately represent the task at hand. While, by now, many methods, e.g., (<xref ref-type="bibr" rid="B129">Reuss et al., 2024a</xref>; <xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B138">Ryu et al., 2024</xref>) only require a low number of five to fifty demonstrations, there are also methods, e.g., (<xref ref-type="bibr" rid="B20">Chen L. et al., 2023</xref>; <xref ref-type="bibr" rid="B140">Saha et al., 2024</xref>) relying on more extensive data sets. Especially offline RL methods, e.g. (<xref ref-type="bibr" rid="B15">Carvalho et al., 2023</xref>; <xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>) usually require extensive amounts of data to accurately predict actions over the complete state-action space, also from suboptimal behavior. Moreover, increasing the variability in training data also has the potential to increase the generalizability of the learned policies. Thus, to automatically increase the variety and size of datasets, without additional costs on researchers and staff, or other more engineering-heavy autonomous data collection pipelines (<xref ref-type="bibr" rid="B187">Yu et al., 2023</xref>), many methodologies, e.g., (<xref ref-type="bibr" rid="B22">Chen Z. et al., 2023</xref>; <xref ref-type="bibr" rid="B93">Mandi et al., 2022</xref>), use DMs for data augmentation. In comparison to other strategies, such as domain randomization (<xref ref-type="bibr" rid="B160">Tremblay et al., 2018</xref>; <xref ref-type="bibr" rid="B159">Tobin et al., 2017</xref>), data augmentation with DMs directly augments the real-world data, making the data grounded in the physical world. In contrast, domain randomization requires complex tuning for each task, to ensure physical plausibility of the randomized scenes, and to enable sim-to-real transfer (<xref ref-type="bibr" rid="B22">Chen Z. et al., 2023</xref>).</p>
<p>Given a set of real-world data, DM-based augmentation methods perform semantically meaningful augmentations via inpainting, such as changing object colors and textures (<xref ref-type="bibr" rid="B198">Zhang X. et al., 2024</xref>), or even replacing whole objects, as well as corresponding language task descriptions (<xref ref-type="bibr" rid="B22">Chen Z. et al., 2023</xref>; <xref ref-type="bibr" rid="B187">Yu et al., 2023</xref>; <xref ref-type="bibr" rid="B93">Mandi et al., 2022</xref>). This enables both the augmentation of objects, which are part of the manipulation process, and backgrounds. The former increases the generalizability to different tasks and objects, while the latter increases robustness to scene information, which should not influence the policy. Some (<xref ref-type="bibr" rid="B198">Zhang X. et al., 2024</xref>) also augment object positions and the corresponding trajectories to generate off-distribution demonstrations for DAgger, thus addressing the covariate shift problem in imitation learning. Others (<xref ref-type="bibr" rid="B21">Chen L. Y. et al., 2024</xref>) augment camera view, robot embodiments, or even (<xref ref-type="bibr" rid="B63">Katara et al., 2024</xref>) generate whole simulation scenes from given URDF files, prompted by a Large Language Model (LLM). Targeted towards offline RL methods, <xref ref-type="bibr" rid="B26">Di Palo et al. (2024)</xref> combines data augmentation with a form of hindsight-experience replay (<xref ref-type="bibr" rid="B3">Andrychowicz et al., 2017</xref>) to adapt the visual observations to the language-task instruction. This increases the number of successful executions in the replay buffer, which potentially increases the data efficiency. The method is used to learn policies for new tasks, on previously collected data, to align the data with the new task instructions.</p>
<p>From a methodological perspective the methods mostly employ frozen web-scale pretrained language (<xref ref-type="bibr" rid="B187">Yu et al., 2023</xref>), and vision-language models, for object segmentation (<xref ref-type="bibr" rid="B187">Yu et al., 2023</xref>), or text-to-image synthesis (Stable Diffusion) (<xref ref-type="bibr" rid="B132">Rombach et al., 2022a</xref>; <xref ref-type="bibr" rid="B93">Mandi et al., 2022</xref>), or finetune (<xref ref-type="bibr" rid="B198">Zhang X. et al., 2024</xref>; <xref ref-type="bibr" rid="B26">Di Palo et al., 2024</xref>) pretrained internet-scale vision-language models. Apart from <xref ref-type="bibr" rid="B198">Zhang X. et al. (2024)</xref> the methods, do not augment actions, but only observations. Thus, the methodologies must ensure augmentations, for which the demonstrated actions do not change, which highly limits the types of augmentations. Moreover, large-scale data scaling via scene augmentation also requires additional computational cost. While this might not be a severe limitation, if it is applied once before the training, it may highly increase training time for online-RL methods.</p>
</sec>
<sec id="s4-3-2">
<label>4.3.2</label>
<title>Sensor data reconstruction</title>
<p>A challenge in vision-based robotic manipulation pertains to the incomplete sensor data. Especially single-view camera setups lead to incomplete object point clouds or images, making accurate grasp and trajectory prediction challenging. This is exacerbated by more complex task settings, with occlusion, as well as inaccurate sensor data.</p>
<p>Multiple methods (<xref ref-type="bibr" rid="B62">Kasahara et al., 2024</xref>; <xref ref-type="bibr" rid="B53">Ikeda et al., 2024</xref>) reconstruct camera viewpoints with DMs. Given an RGBD image and camera intrinsics <xref ref-type="bibr" rid="B62">Kasahara et al. (2024)</xref> generates new object views without requiring CAD models of the objects. For this, the existing points are projected to the new viewpoint. The scene is segmented using the vision foundation model SAM (<xref ref-type="bibr" rid="B68">Kirillov et al., 2023</xref>), to create object masks. On these masks missing data points are inpainted using the pretrained diffusion model for image generation Dall<inline-formula id="inf124">
<mml:math id="m133">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>E (<xref ref-type="bibr" rid="B60">Kapelyukh et al., 2023</xref>). As Dall<inline-formula id="inf125">
<mml:math id="m134">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>E does not ensure spatial consistency, consistency filtering is applied across viewpoints. Moreover, Dall<inline-formula id="inf126">
<mml:math id="m135">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>E, only processes 2D images. Thus, to also complete the missing depth information, a model is trained to predict the missing depth information from the projected depth map and the reconstructed image. In this method the viewpoints are sampled on evenly spaced directions along a viewing sphere. However, generating the point clouds for many viewpoints is computationally expensive, and might not be necessary for successful task completion. Thus, view-planning is applied to generate a minimal set of views <xref ref-type="bibr" rid="B114">Pan S. et al. (2024)</xref>, <xref ref-type="bibr" rid="B115">Pan et al. (2025)</xref>. use a DM to generate geometric priors from a 2D image, enabling a view-planner to sample a minimum set of viewpoints that minimize movement cost. The views are then used to train a Neural Radiance Field (NeRF) (<xref ref-type="bibr" rid="B102">Mildenhall et al., 2020</xref>) to reconstruct 3D scenes from 2D images.</p>
<p>In the field of robotic manipulation, not many methods consider scene reconstruction. A possible reason for this is its relatively high computational cost. However, expanding to the areas of robotics and computer vision, more methodologies in the field of scene reconstruction exist. In robotic manipulation instead more methods focus on making policies more robust to incomplete or noisy sensor information, e.g., (<xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>; <xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>). However, the limited number of occlusion in the experimental setups indicate that strong occlusion are still a major challenge. Moreover, scene reconstruction is unable to react to completely occluded objects.</p>
</sec>
<sec id="s4-3-3">
<label>4.3.3</label>
<title>Object rearrangement</title>
<p>The ability of DMs for text-to-image synthesis offers the possibility to generate plans from high-level task descriptions. In particular, given an initial visual observation, one group of methods uses such models to generate target-arrangement of objects in the scene, specified by a language-prompt (<xref ref-type="bibr" rid="B86">Liu et al., 2023b</xref>; <xref ref-type="bibr" rid="B60">Kapelyukh et al., 2023</xref>; <xref ref-type="bibr" rid="B181">Xu et al., 2024</xref>; <xref ref-type="bibr" rid="B191">Zeng et al., 2024</xref>; <xref ref-type="bibr" rid="B59">Kapelyukh et al., 2024</xref>). Examples of applications could be setting up a dinner table or clearing up a kitchen counter. While the earlier methodologies (<xref ref-type="bibr" rid="B60">Kapelyukh et al., 2023</xref>; <xref ref-type="bibr" rid="B86">Liu et al., 2023b</xref>) use the pretrained VLM Dall<inline-formula id="inf127">
<mml:math id="m136">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>E (<xref ref-type="bibr" rid="B8">Black et al., 2024b</xref>) to generate rearrangements in a zero-shot manner, this has the disadvantage of possibly introducing scene inconsistencies and incompatibilities, due to the lack of geometric understanding and object permanence. Thus, the later methods (<xref ref-type="bibr" rid="B181">Xu et al., 2024</xref>; <xref ref-type="bibr" rid="B59">Kapelyukh et al., 2024</xref>) use combinations of pretrained LLMs and VLMs like CLIP (<xref ref-type="bibr" rid="B98">Meila and Zhang, 2021</xref>), together with other non-diffusion visual processing methods like NeRF (<xref ref-type="bibr" rid="B102">Mildenhall et al., 2020</xref>) and SAM (<xref ref-type="bibr" rid="B68">Kirillov et al., 2023</xref>), and custom DMs. The described methodologies are similar to the methods for object pose diffusion (<xref ref-type="bibr" rid="B103">Mishra and Chen, 2024</xref>; <xref ref-type="bibr" rid="B149">Simeonov et al., 2023</xref>; <xref ref-type="bibr" rid="B203">Zhao et al., 2025</xref>) mentioned in <xref ref-type="sec" rid="s4-2">Section 4.2</xref>. The main difference is that the methods here focus on the rearrangement of multiple objects specified by a sparse language input, not exhaustively describing the geometric layout of the target arrangement. Different to the methods from <xref ref-type="sec" rid="s4-2">Section 4.2</xref>, the integration with grasp or motion planning to achieve the target arrangement is not the focus. However, nonetheless for all of the above listed methodologies for object rearrangement their effectiveness is also demonstrated in real-robot experiments.</p>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Experiments and benchmarks</title>
<p>In this section, we focus on the evaluation of the various DMs for robotic manipulation. Details on the employed benchmarks and baselines are listed in the separate tables for imitation learning (<xref ref-type="table" rid="T6">Table 6</xref>), reinforcement learning (<xref ref-type="table" rid="T7">Table 7</xref>) in the Appendix, and grasp learning (<xref ref-type="table" rid="T5">Table 5</xref>). Separately, the references for all applied benchmarks are listed in <xref ref-type="sec" rid="s12">Supplementary Appendix Table 2</xref>.</p>
<table-wrap id="T6" position="float">
<label>TABLE 6</label>
<caption>
<p>Benchmarks of trajectory diffusion using reinforcement learning.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Reference</th>
<th rowspan="2" align="center">Diffusion Baseline</th>
<th colspan="2" align="center">Simulation</th>
<th colspan="2" align="center">Real World</th>
</tr>
<tr>
<th align="center">Benchmark</th>
<th align="center">&#x23;Demos</th>
<th align="center">Real</th>
<th align="center">&#x23;Demos</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Diffuser (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>)</td>
<td align="left">&#x2717;</td>
<td align="left">KUKA (custom)</td>
<td align="left">10 k</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">Decision Diffuser (<xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>)</td>
<td align="left">&#x2717;</td>
<td align="left">D4RLKitchen, KUKA</td>
<td align="left">/, 10 k</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">Diffusion-QL (<xref ref-type="bibr" rid="B170">Wang et al., 2023b</xref>)</td>
<td align="left">&#x2717;</td>
<td align="left">D4RLKitchen</td>
<td align="left">10000 trans &#x2a;<sup>1</sup>
</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B170">Wang et al. (2023b)</xref>
</td>
<td align="left">Diffuser</td>
<td align="left">custom</td>
<td align="left">8 k</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">HDMI (<xref ref-type="bibr" rid="B75">Li et al., 2023</xref>)</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
<td align="left">-</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B25">Ding and Jin (2023)</xref>
</td>
<td align="left">Diffusion-QL</td>
<td align="left">D4RLKitchen, Adroit</td>
<td align="left">/</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B104">Mishra et al. (2023)</xref>
</td>
<td align="left">Decision Diffuser</td>
<td align="left">STAP</td>
<td align="left">/</td>
<td align="center">&#x2713;</td>
<td align="left">/</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B58">Kang et al. (2023)</xref>
</td>
<td align="left">Diffusion-QL</td>
<td align="left">Adroit, D4RL Kitchen</td>
<td align="left">/</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B11">Brehmer et al. (2023)</xref>
</td>
<td align="left">Diffuser</td>
<td align="left">KUKA</td>
<td align="left">/</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B156">Suh et al. (2023)</xref>
</td>
<td align="left">Diffuser</td>
<td align="left">&#x2713;</td>
<td align="left">-</td>
<td align="center">&#x2713;</td>
<td align="left">100</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B43">Ha et al. (2023)</xref>
</td>
<td align="left">Diffusion-QL</td>
<td align="left">&#x2717;</td>
<td align="left">-</td>
<td align="center">&#x2713;</td>
<td align="left">50</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B66">Kim et al. (2024b)</xref>
</td>
<td align="left">Diffuser, Decision Diffuser, HDMI</td>
<td align="left">Fetch env</td>
<td align="left">/</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B79">Liang et al. (2023)</xref>
</td>
<td align="left">Diffuser, Decision Diffuser</td>
<td align="left">KUKA</td>
<td align="left">/</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B192">Zhang et al. (2024a)</xref>
</td>
<td align="left">Diffuser</td>
<td align="left">CALVIN, CLEVR-Robot</td>
<td align="left">/</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B1">Ada et al. (2024)</xref>
</td>
<td align="left">Diffusion-QL</td>
<td align="left">&#x2713;</td>
<td align="left">/</td>
<td align="center">&#x2713;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B128">Ren et al. (2024)</xref>
</td>
<td align="left">Diffusion-QL</td>
<td align="left">Robomimic, D3IL, FurnitureBench</td>
<td align="left">100-300, 96,50</td>
<td align="center">&#x2713; &#x2a;<sup>2</sup>
</td>
<td align="left">50</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B50">Huang et al. (2025b)</xref>
</td>
<td align="left">Diffusion-QL</td>
<td align="left">MetaWorld, Adroit</td>
<td align="left">20, 50</td>
<td align="center">&#x2713;</td>
<td align="left">50</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B15">Carvalho et al. (2023)</xref>
</td>
<td align="left">&#x2717;</td>
<td align="left">custom</td>
<td align="left">25</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>For each benchmark, the numbers of demonstrations are listed in the same order. In the column &#x201c;Diffusion Baselines&#x201d; only those baselines, which are diffusion methods themselves, are listed. Methods not evaluated against a diffusion-based baseline, indicated by an (&#x2717;), are only evaluated against non-diffusion baselines or ablations of the method. The references for the benchmarks are listed in <xref ref-type="sec" rid="s12">Supplementary Appendix Table 2</xref>. In the following, the symbols are explained: &#x2a;<sup>1</sup> As the number refers to the number of transitions, not demonstrations, this high number is expected. A (&#x2713;) in the column &#x201c;Benchmark&#x201d; indicates that the method is evaluated in simulation, but not with a robotic manipulation task, while a (&#x2717;) indicates that the method is not evaluated in simulation. The column &#x201c;Real&#x201d; indicates whether methods are evaluated in the real world (&#x2713;), or not (&#x2717;). A &#x201c;/&#x201d; indicates that the information is not provided by the cited paper, while a &#x201d;-&#x201d; indicates that the information does not apply.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<table-wrap id="T7" position="float">
<label>TABLE 7</label>
<caption>
<p>Benchmarks of trajectory diffusion using imitation learning.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th rowspan="2" align="center">Reference</th>
<th rowspan="2" align="center">Diffusion Baseline</th>
<th colspan="2" align="center">Simulation</th>
<th colspan="2" align="center">Real World</th>
</tr>
<tr>
<th align="center">Benchmark</th>
<th align="center">&#x23;Demos</th>
<th align="center">Real</th>
<th align="center">&#x23;Demos</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Diffusion Policy (DP) (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>)</td>
<td align="left">&#x2717;</td>
<td align="left">FrankaKitchen, Robomimic, custom</td>
<td align="left">566,500, /</td>
<td align="center">&#x2713;</td>
<td align="left">/</td>
</tr>
<tr>
<td align="left">ChainedDiffuser (<xref ref-type="bibr" rid="B179">Xian et al., 2023</xref>)</td>
<td align="left">&#x2717;</td>
<td align="left">RLBench</td>
<td align="left">100</td>
<td align="center">&#x2713;</td>
<td align="left">10 - 20</td>
</tr>
<tr>
<td align="left">BESO (<xref ref-type="bibr" rid="B130">Reuss et al., 2023</xref>)</td>
<td align="left">DP, Diffusion-BC</td>
<td align="left">Relay Kitchen, CALVIN, custom</td>
<td align="left">566, /,1000</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B20">Chen et al. (2023a)</xref>
</td>
<td align="left">&#x2717;</td>
<td align="left">CALVIN, FrankaKitchen, Ravens</td>
<td align="left">200K, 566, 1000</td>
<td align="center">&#x2713;</td>
<td align="left">/</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B208">Zhou et al. (2023)</xref>
</td>
<td align="left">
<inline-formula id="inf132">
<mml:math id="m141">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>Diffuser</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>,Decision <inline-formula id="inf133">
<mml:math id="m142">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>Diffuser</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">RLBench</td>
<td align="left"/>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">Diffusion-BC (<xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>)</td>
<td align="left">&#x2717;</td>
<td align="left">D4RLKitchen</td>
<td align="left">566</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B99">Mendez-Mendez et al. (2023)</xref>
</td>
<td align="left">&#x2717;</td>
<td align="left">BEHAVIOR</td>
<td align="left">-</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">3D-DP (<xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>)</td>
<td align="left">DP</td>
<td align="left">e.g. Adroit, MetaWorld, DexDeform</td>
<td align="left">10 - 100</td>
<td align="center">&#x2713;</td>
<td align="left">40</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B64">Ke et al. (2024)</xref>
</td>
<td align="left">3D-DP, ChainedDiffuser</td>
<td align="left">RLBench, CALVIN</td>
<td align="left">24 h</td>
<td align="center">&#x2713;</td>
<td align="left">15</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B87">Liu et al. (2023c)</xref>
</td>
<td align="left">DP, SE(3)-DM</td>
<td align="left">Custom</td>
<td align="left">/</td>
<td align="center">&#x2713;</td>
<td align="left">/</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B121">Power et al. (2023)</xref>
</td>
<td align="left">&#x2717;</td>
<td align="left">custom</td>
<td align="left">/</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B92">Ma et al. (2024b)</xref>
</td>
<td align="left">DP, Diffuser</td>
<td align="left">RLBench</td>
<td align="left">100</td>
<td align="center">&#x2713;</td>
<td align="left">20</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B164">Vosylius et al. (2024)</xref>
</td>
<td align="left">DP</td>
<td align="left">RLBench</td>
<td align="left">20</td>
<td align="center">&#x2713;</td>
<td align="left">20</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B192">Zhang et al. (2024a)</xref>
</td>
<td align="left">Diffuser</td>
<td align="left">CALVIN</td>
<td align="left">/</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B129">Reuss et al. (2024a)</xref>
</td>
<td align="left">&#x2717;</td>
<td align="left">CALVIN, LIBERO</td>
<td align="left">24 h, 50</td>
<td align="center">&#x2713;</td>
<td align="left">4.5 h</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B142">Scheikl et al. (2024)</xref>
</td>
<td align="left">DP, BESO</td>
<td align="left">LapGym</td>
<td align="left">90 - 200</td>
<td align="center">&#x2713;</td>
<td align="left">90 - 200</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B19">Chen et al. (2024a)</xref>
</td>
<td align="left">DP, SE(3)-DM</td>
<td align="left">FrankaKitchen, Adroit</td>
<td align="left">16k - 64k, 1.25k - 5 k</td>
<td align="center">&#x2713;</td>
<td align="left">60</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B206">Zhou et al. (2024a)</xref>
</td>
<td align="left">DP, BESO, Consistency <inline-formula id="inf134">
<mml:math id="m143">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>Models</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="left">Relay Kitchen, XArm Block Push, D3IL</td>
<td align="left">566, 1k,96 - 2 k</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B76">Li et al. (2024c)</xref>
</td>
<td align="left">DP, 3D-DP</td>
<td align="left">Robomimic, custom</td>
<td align="left">500,100</td>
<td align="center">&#x2713;</td>
<td align="left">100</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B148">Si et al. (2024)</xref>
</td>
<td align="left">DP</td>
<td align="left">&#x2717;</td>
<td align="left">-</td>
<td align="center">&#x2713;</td>
<td align="left">25 - 50</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B140">Saha et al. (2024)</xref>
</td>
<td align="left">&#x2717;</td>
<td align="left">M<inline-formula id="inf135">
<mml:math id="m144">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>Nets</td>
<td align="left">6.54Mil</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B6">Bharadhwaj et al. (2024b)</xref>
</td>
<td align="left">&#x2717;</td>
<td align="left">EpicKitchens, RT1, BridgeData</td>
<td align="left">400 <inline-formula id="inf536">
<mml:math id="m545">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>k</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">&#x2713;</td>
<td align="left">400</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B167">Wang et al. (2024b)</xref>
</td>
<td align="left">&#x2717;</td>
<td align="left">custom</td>
<td align="left">50 K <inline-formula id="inf136">
<mml:math id="m145">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>trans</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center">&#x2713;</td>
<td align="left">50 K <inline-formula id="inf137">
<mml:math id="m146">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mtext>trans</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B72">Li et al. (2025)</xref>
</td>
<td align="left">DP, 3D-DP</td>
<td align="left">RLBench</td>
<td align="left">40</td>
<td align="center">&#x2713;</td>
<td align="left">40</td>
</tr>
<tr>
<td align="left">
<xref ref-type="bibr" rid="B131">Reuss et al. (2024b)</xref>
</td>
<td align="left">DP</td>
<td align="left">CALVIN, LIBERO, Relay Kitchen, Block Push</td>
<td align="left">22966, 50, 566, 1000</td>
<td align="center">&#x2717;</td>
<td align="left">-</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>For each benchmark, the numbers of demonstrations are listed in the same order. In the column &#x201c;Diffusion Baselines&#x201d; only those baselines, which are diffusion methods themselves, are listed. Methods not evaluated against a diffusion-based baseline, indicated by an (&#x2717;), are only evaluated against non-diffusion baselines or ablations of the method. The references for the benchmarks are listed in <xref ref-type="sec" rid="s12">Supplementary Appendix Table 2</xref>. In the following, he symbols are explained: Methods by and &#x2a;<sup>1</sup> <xref ref-type="bibr" rid="B55">Janner et al. (2022)</xref>, and &#x2a;<sup>2</sup> <xref ref-type="bibr" rid="B2">Ajay et al. (2023)</xref>, and &#x2a;<sup>3</sup> (<xref ref-type="bibr" rid="B153">Song et al., 2024</xref>). &#x2a;<sup>4</sup> The diffusion model is trained using uncurated video data. &#x2a;<sup>5</sup> As the number refers to the number of transitions, not demonstrations, this high number is expected. The column &#x201c;Real&#x201d; indicates whether methods are evaluated in the real world (&#x2713;), or not (&#x2717;). A &#x201c;/&#x201d; indicates that the information is not provided by the cited paper, while a &#x201d;-&#x201d; indicates that the information does not apply.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Various benchmarks are used to evaluate the methods. Common benchmarks are CALVIN (<xref ref-type="bibr" rid="B97">Mees et al., 2022</xref>), RLBench (<xref ref-type="bibr" rid="B54">James et al., 2020</xref>), RelayKitchen (<xref ref-type="bibr" rid="B42">Gupta et al., 2020</xref>), and Meta-World (<xref ref-type="bibr" rid="B186">Yu et al., 2020</xref>). Primarily in RL, the benchmark D4RL Kitchen (<xref ref-type="bibr" rid="B35">Fu et al., 2020</xref>) is used. One method (<xref ref-type="bibr" rid="B128">Ren et al., 2024</xref>) uses FurnitureBench (Heo et al., 0) for real-world manipulation tasks. Adroit (<xref ref-type="bibr" rid="B126">Rajeswaran et al., 2017</xref>) is a common benchmark for dexterous manipulation, LIBERO (<xref ref-type="bibr" rid="B83">Liu B. et al., 2023</xref>) for lifelong learning, and LapGym (<xref ref-type="bibr" rid="B94">Maria Scheikl et al., 2023</xref>) for medical tasks.</p>
<p>Many methods are only being evaluated against baselines, which are not based on DMs themselves. However, there are some common DM-based baselines. For methods operating in <inline-formula id="inf128">
<mml:math id="m137">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>-space (<xref ref-type="bibr" rid="B19">Chen K. et al., 2024</xref>; <xref ref-type="bibr" rid="B153">Song et al., 2024</xref>; <xref ref-type="bibr" rid="B138">Ryu et al., 2024</xref>), <inline-formula id="inf129">
<mml:math id="m138">
<mml:mrow>
<mml:mi mathvariant="bold">S</mml:mi>
<mml:mi mathvariant="bold">E</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>-Diffusion Policy (<xref ref-type="bibr" rid="B162">Urain et al., 2023</xref>), probably the first paper using DMs for grasp generation, is commonly used as baseline. For RL-based methods, the RL-based Diffuser (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>), Diffusion-QL (<xref ref-type="bibr" rid="B169">Wang et al., 2023a</xref>), and Decision Diffuser (<xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>) are commonly used as baselines. It should be noted that in the original paper, Decision Diffuser (<xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>) is evaluated against Diffuser (<xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>) and outperforms it on almost all tasks, particularly on the manipulation tasks, block stacking, and rearrangement. However, neither of these methods is evaluated on real-world tasks. Another common baseline is DP (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>), as many methods are developed based on it. A common baseline for methods integrating 3D visual representations is 3D Diffusion Policy (<xref ref-type="bibr" rid="B190">Ze et al., 2024</xref>). 3D Diffusion Policy is evaluated against DP, and outperforms it on a huge variety of tasks in the benchmarks Adroit, MetaWorld, and Dexart with an average success rate of 74.4%, outperforming DP by 24.2%. It is also evaluated on four real-world manipulation tasks: rolling and pinching a dumpling, drilling, and pouring. With an average success rate of 85.0% it outperforms DP by 50%. 3D Diffusion Policy is greatly outperformed by 3D Diffuser Actor (<xref ref-type="bibr" rid="B64">Ke et al., 2024</xref>) on the CALVIN benchmark, especially for zero-shot long-horizon tasks. However, no comparison for real-world tasks is provided.</p>
<p>The majority of methods are evaluated in simulation as well as in real-world experiments. For real-world experiments, most policies are directly trained on real-world data. However, some are trained exclusively in simulation and applied in the real world in a zero shot (<xref ref-type="bibr" rid="B187">Yu et al., 2023</xref>; <xref ref-type="bibr" rid="B104">Mishra et al., 2023</xref>; <xref ref-type="bibr" rid="B128">Ren et al., 2024</xref>; <xref ref-type="bibr" rid="B86">Liu et al., 2023b</xref>; <xref ref-type="bibr" rid="B59">Kapelyukh et al., 2024</xref>; <xref ref-type="bibr" rid="B87">Liu et al., 2023c</xref>), utilizing domain randomization, or real-world scene reconstruction in simulation. Few, predominately RL methods, are only evaluated in simulation (<xref ref-type="bibr" rid="B183">Yang et al., 2023</xref>; <xref ref-type="bibr" rid="B121">Power et al., 2023</xref>; <xref ref-type="bibr" rid="B169">Wang et al., 2023a</xref>; <xref ref-type="bibr" rid="B55">Janner et al., 2022</xref>; <xref ref-type="bibr" rid="B116">Pearce et al., 2022</xref>; <xref ref-type="bibr" rid="B170">Wang et al., 2023b</xref>; <xref ref-type="bibr" rid="B99">Mendez-Mendez et al., 2023</xref>; <xref ref-type="bibr" rid="B66">Kim S. et al., 2024</xref>; <xref ref-type="bibr" rid="B11">Brehmer et al., 2023</xref>; <xref ref-type="bibr" rid="B79">Liang et al., 2023</xref>; <xref ref-type="bibr" rid="B206">Zhou H. et al., 2024</xref>; <xref ref-type="bibr" rid="B103">Mishra and Chen, 2024</xref>; <xref ref-type="bibr" rid="B2">Ajay et al., 2023</xref>; <xref ref-type="bibr" rid="B25">Ding and Jin, 2023</xref>; <xref ref-type="bibr" rid="B192">Zhang E. et al., 2024</xref>).</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Conclusion, limitations and outlook</title>
<p>Diffusion models (DMs) have emerged as state-of-the-art methods in robotic manipulation, offering exceptional ability in modeling multi-modal distributions, high training stability, and stability to high-dimensional input and output spaces. Several tasks, challenges, and limitations in the domain of robotic manipulation with DMs remain unsolved. A prevalent issue is the lack of generalizability. The slow inference time for DMs also remains a major bottleneck.</p>
<sec id="s6-1">
<label>6.1</label>
<title>Limitations</title>
<sec id="s6-1-1">
<label>6.1.1</label>
<title>Generalizability</title>
<p>While a lot of methods demonstrate relatively good generalizability in terms of object types, lightning conditions, and task complexity, they still face limitations in this area. This prevalent limitation shared across almost all methodologies in robotic manipulation remains a crucial research question.</p>
<p>The majority of methods using DMs for trajectory generation rely on imitation learning, using mostly behavior cloning. Thus, they inherit the dependence on the quality and diversity of training data, making it difficult to handle out-of-distribution situations due to the covariate shift problem (<xref ref-type="bibr" rid="B136">Ross and Bagnell, 2010</xref>). As most methodologies combining DMs with RL use offline RL, they still rely on existing data, mapping a sufficient amount of the state-action space, and are thus also unable to react to distribution shifts. Moreover, offline RL requires more careful fine-tuning than imitation learning to ensure training stability and prevent overfitting. Still, the advantage of RL is that it can handle suboptimal behavior <xref ref-type="bibr" rid="B71">Levine et al. (2020)</xref>.</p>
<p>While data scaling offers improved generalizability, it typically demands large training datasets and substantial computational resources. One recent solution is to use pre-trained foundation models. Moreover, as the majority of current methods for data augmentation in DMs do not augment trajectories, e.g., (<xref ref-type="bibr" rid="B187">Yu et al., 2023</xref>; <xref ref-type="bibr" rid="B93">Mandi et al., 2022</xref>), it only increases robustness to slightly different task settings, such as changes in colors, textures, distractors, and background. VLAs can generalize to multi-task and long-horizon settings but often lack action precision, thus requiring finetuning and the combination with more specialized agents (<xref ref-type="bibr" rid="B201">Zhang et al., 2024g</xref>).</p>
</sec>
<sec id="s6-1-2">
<label>6.1.2</label>
<title>Sampling speed</title>
<p>The principal limitation inherent to DMs can be attributed to the iterative nature of the sampling process, which results in a time-intensive sampling procedure, thus impeding efficiency and real-time prediction capabilities. Despite recent advances that improve sampling speed and quality (<xref ref-type="bibr" rid="B19">Chen K. et al., 2024</xref>; <xref ref-type="bibr" rid="B206">Zhou H. et al., 2024</xref>), a considerable number of recent methods use DDIM (<xref ref-type="bibr" rid="B152">Song J. et al., 2021</xref>), although other methods, such as DPM-solver (<xref ref-type="bibr" rid="B88">Lu et al., 2022</xref>) have shown better performance. However, this comparison has only been performed using image generation benchmarks and would need to be verified for applications in robotic manipulation. Numerous works have demonstrated competitive task performance using DDIM, but do not directly investigate the decrease in task performance associated with a lower number of reverse diffusion steps. <xref ref-type="bibr" rid="B69">Ko et al. (2024)</xref> analyzes their approach using both DDPM and DDIM sampling, reporting a sampling process that is ten times faster with only a 5.6% decrease in task performance when using DDIM. Although such a decline might appear negligible, its significance is highly task-dependent. Consequently, there is a need for efficient sampling strategies and a more comprehensive analysis of existing sampling methods, particularly regarding the domain of robotic manipulation. It should, however, be noted that already in DP (<xref ref-type="bibr" rid="B23">Chi et al., 2023</xref>), one of the earlier methods combining DMs with receding-horizon control for trajectory planning, real-time control is possible. Using DDIM with 10 denoising steps during inference, they report an inference latency of 0.1 s on a Nvidia 3080 GPU.</p>
</sec>
</sec>
<sec id="s6-2">
<label>6.2</label>
<title>Conclusion and outlook</title>
<p>This survey, to the best to our knowledge, is the first survey reviewing the state-of-the-art methods diffusion models (DMs) in robotics manipulation. This paper offers a thorough discussion of various methodologies regarding network architecture, learning framework, application, and evaluation, highlighting limits and advantages. We explored the three primary applications of DMs in robotic manipulation: trajectory generation, robotic grasping, and visual data augmentation. Most notably, DMs offer exceptional ability in modeling multi-modal distributions, high training stability, and robustness to high-dimensional input and output spaces. Especially in visual robotic manipulation, DMs provide essential capabilities to process high-resolution 2D and 3D visual observations, as well as to predict high-dimensional trajectories and grasp poses, even directly in image space.</p>
<p>A key challenge of DMs is the slow inference speed. In the field of computer vision, fast samplers have been developed that have not yet been evaluated in the field of robotic manipulation. Testing those samplers and comparing them against the commonly used ones, could be one step to increase sampling efficiency. Moreover, there are also methods for fast sampling, specifically in robotic manipulation, that are not broadly used, e.g. BRIDGeR (<xref ref-type="bibr" rid="B19">Chen K. et al., 2024</xref>). While the generalizability of DMs remains also an open challenge, the image generation capabilities of DMs open new avenues in data augmentation for data scaling, making methods more robust to limited data variety. Generalizability could be also improved by the integration of advanced vision-language, and vision-language action models.</p>
<p>We believe continual learning could be a promising approach to improve generalizability and adaptability in highly dynamic and unfamiliar environments. This remains a widely unexplored problem domain for DMs in robotic manipulation, exceptions are (<xref ref-type="bibr" rid="B26">Di Palo et al., 2024</xref>; <xref ref-type="bibr" rid="B99">Mendez-Mendez et al., 2023</xref>). However, these methods have strong limitations. For instance, <xref ref-type="bibr" rid="B26">Di Palo et al. (2024)</xref> relies on precise feature descriptions of all involved objects and is restricted to predefined abstract skills. Moreover, their continual update process involves replaying all past data, which is both computationally inefficient and does not prevent catastrophic forgetting. Morover, to handle complex and cluttered scenes, view planning and iterative planning strategies, also considering complete occlusions, could be combined with existing DMs using 3D scene representations. Leveraging the semantic reasoning capabilities of vision language and vision language action models could be a possible approach.</p>
</sec>
</sec>
</body>
<back>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>RW: Writing &#x2013; original draft, Writing &#x2013; review and editing. YS: Writing &#x2013; review and editing, Writing &#x2013; original draft. SL: Writing &#x2013; original draft. RR: Writing &#x2013; original draft, Supervision, Writing &#x2013; review and editing.</p>
</sec>
<ack>
<title>Acknowledgements</title>
<p>We thank our colleague Edgar Welte for providing the video data for the illustration of the diffusion process in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
</ack>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frobt.2025.1606247/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frobt.2025.1606247/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn fn-type="custom" custom-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/429733/overview">David Howard</ext-link>, Commonwealth Scientific and Industrial Research Organisation (CSIRO), Australia</p>
</fn>
<fn fn-type="custom" custom-type="reviewed-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1381431/overview">Fan Zhang</ext-link>, Imperial College London, United Kingdom</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3066640/overview">Ananth Jonnavittula</ext-link>, Virginia Tech, United States</p>
</fn>
<fn id="n1">
<label>1</label>
<p>In the context of probability distributions, &#x201c;multi-modal&#x201d; does not refer to multiple input modalities but rather to the presence of multiple peaks (modes) in the distribution, each representing a distinct possible outcome. For example, in trajectory planning, a multi-modal distribution can capture multiple feasible trajectories. Accurately modeling all modes is crucial for policies, as it enables better generalization to diverse scenarios during inference</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ada</surname>
<given-names>S. E.</given-names>
</name>
<name>
<surname>Oztop</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ugur</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Diffusion policies for out-of-distribution generalization in offline reinforcement learning</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>9</volume>, <fpage>3116</fpage>&#x2013;<lpage>3123</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2024.3363530</pub-id>
</mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ajay</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jaakkola</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>IS conditional generative modeling all you need for decision-making?</article-title>,&#x201d; in <conf-name>The Eleventh International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Andrychowicz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Wolski</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ray</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Welinder</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Hindsight experience replay</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>30</volume>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2017">https://proceedings.neurips.cc/paper_files/paper/2017</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barad</surname>
<given-names>K. R.</given-names>
</name>
<name>
<surname>Orsula</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Richard</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dentler</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Olivares-Mendez</surname>
<given-names>M. A.</given-names>
</name>
<name>
<surname>Martinez</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>GraspLDM: generative 6-DoF grasp synthesis using latent diffusion models</article-title>. <source>IEEE Access</source> <volume>12</volume>, <fpage>164621</fpage>&#x2013;<lpage>164633</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2024.3492118</pub-id>
</mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bharadhwaj</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Tulsiani</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024a</year>). &#x201c;<article-title>Towards generalizable zero-shot manipulation via translating human interaction plans</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>6904</fpage>&#x2013;<lpage>6911</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10610288</pub-id>
</mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Bharadhwaj</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Mottaghi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tulsiani</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024b</year>). &#x201c;<article-title>Track2Act: predicting point tracks from internet videos enables generalizable robot manipulation</article-title>,&#x201d; in <conf-name>1st Workshop on X-Embodiment Robot Learning</conf-name>, <fpage>306</fpage>&#x2013;<lpage>324</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-73116-7_18</pub-id>
</mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Black</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Driess</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Esmail</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Equi</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). <source>&#x3c0;<sub>0</sub>: a vision-language-action flow model for general robot control</source>. <comment>arXiv preprint arXiv:2410.24164</comment>.</mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Black</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Nakamoto</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Atreya</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Walke</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2024b</year>). &#x201c;<article-title>Zero-shot robotic manipulation with pretrained image-editing diffusion models</article-title>,&#x201d; in <conf-name>12th International Conference on Learning Representations, ICLR 2024</conf-name>.</mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bohg</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Morales</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Asfour</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kragic</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Data-driven grasp synthesis&#x2014;a survey</article-title>. <source>IEEE Trans. robotics</source> <volume>30</volume>, <fpage>289</fpage>&#x2013;<lpage>309</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2013.2289018</pub-id>
</mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Braun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jaquier</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Rozo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Asfour</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Riemannian flow matching policy for robot motion learning</article-title>,&#x201d; in <conf-name>2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>, <fpage>5144</fpage>&#x2013;<lpage>5151</lpage>. <pub-id pub-id-type="doi">10.1109/IROS58592.2024.10801521</pub-id>
</mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brehmer</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Bose</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>de Haan</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Cohen</surname>
<given-names>T. S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>EDGI: equivariant diffusion for planning with embodied agents</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>, <fpage>63818</fpage>&#x2013;<lpage>63834</lpage>. <comment> Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2023">https://proceedings.neurips.cc/paper_files/paper/2023</ext-link>.</comment> </mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Brohan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Carbajal</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chebotar</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Choromanski</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2023a</year>). <source>RT-2: vision-Language-action models transfer web knowledge to robotic control</source>. <comment>arXiv preprint arXiv:2307.15818</comment>.</mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Brohan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Brown</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Carbajal</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chebotar</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dabis</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2023b</year>). <source>RT-1: robotics transformer for real-world control at scale</source>. <comment>arXiv preprint arXiv:2212.06817</comment>.</mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Cao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kitani</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Multi-modal diffusion for hand-object grasp generation</source>. <comment>arXiv preprint arXiv:2409.04560</comment>.</mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Carvalho</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Baierl</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Koert</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Motion planning diffusion: learning and planning of robot motions with diffusion models</article-title>,&#x201d; in <conf-name>2023 IEEE International Conference on Intelligent Robots and Systems</conf-name>, <fpage>1916</fpage>&#x2013;<lpage>1923</lpage>. <pub-id pub-id-type="doi">10.1109/IROS55552.2023.10342382</pub-id>
</mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Carvalho</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Jahr</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Urain</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Koert</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <source>Grasp diffusion network: learning grasp generators from partial point clouds with diffusion models in SO (3) xR3</source>. <comment>arXiv preprint arXiv:2412.08398</comment>.</mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Text2Grasp: grasp synthesis by text prompts of object grasping parts</source>. <comment>arXiv preprint arXiv:2404.15189</comment>.</mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Cheang</surname>
<given-names>C.-L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Jing</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <source>GR-2: a generative video-language-action model with web-scale knowledge for robot manipulation</source>. <comment>arXiv preprint arXiv:2410.06158</comment>.</mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lim</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Soh</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2024a</year>). <article-title>Don&#x2019;t start from scratch: behavioral refinement via interpolant-based policy diffusion</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.48550/arXiv.2402.16075</pub-id>
</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bahl</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Pathak</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023a</year>). <article-title>PlayFusion: Skill acquisition via diffusion from language-annotated play</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>2012</fpage>&#x2013;<lpage>2029</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/chen23c.html">https://proceedings.mlr.press/v229/chen23c.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>L. Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Dharmarajan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Irshad</surname>
<given-names>M. Z.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Keutzer</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2024b</year>). &#x201c;<article-title>Rovi-aug: robot and viewpoint augmentation for cross-embodiment robot learning</article-title>,&#x201d; in <conf-name>Conference on Robot Learning (CoRL)</conf-name>.</mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Kiami</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2023b</year>). <source>GenAug: retargeting behaviors to unseen situations via generative augmentation</source>. <comment>arXiv preprint arXiv:2302.06671</comment>.</mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cousineau</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Burchfiel</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Diffusion policy: visuomotor policy learning via action diffusion</article-title>. <source>Robotics Sci. Syst. (RSS)</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2303.04137</pub-id>
</mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Nichol</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Diffusion models beat GANs on image synthesis</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>34</volume>, <fpage>8780</fpage>&#x2013;<lpage>8794</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2021">https://proceedings.neurips.cc/paper_files/paper/2021</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Consistency models as a rich and efficient policy class for reinforcement learning</article-title>,&#x201d; in <conf-name>International Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Di Palo</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Hasenclever</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Humplik</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Byravan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Diffusion augmented agents: a framework for efficient exploration and transfer learning</source>. <comment>arXiv preprint arXiv:2407.20798</comment>.</mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Dosovitskiy</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Beyer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Kolesnikov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Weissenborn</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Unterthiner</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>An image is worth 16x16 words: transformers for image recognition at scale</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nachum</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Learning universal policies via text-guided video generation</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>, <fpage>9156</fpage>&#x2013;<lpage>9172</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2023">https://proceedings.neurips.cc/paper_files/paper/2023</ext-link>
</comment>
</mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fang</surname>
<given-names>H.-S.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gou</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Graspnet-1billion: a large-scale benchmark for general object grasping</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>, <fpage>11444</fpage>&#x2013;<lpage>11453</lpage>.</mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Feng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Triebel</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Knoll</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024</year>). <source>FFHFlow: a flow-based variational approach for multi-fingered grasp synthesis in real time</source>. <comment>arXiv preprint arXiv:2407.15161</comment>.</mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Firoozi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Tucker</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Majumdar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Foundation models in robotics: applications, challenges, and the future</article-title>. <source>Int. J. Robotics Res.</source> <volume>44</volume>, <fpage>701</fpage>&#x2013;<lpage>739</lpage>. <pub-id pub-id-type="doi">10.1177/02783649241281508</pub-id>
</mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Florence</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lynch</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ramirez</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Wahid</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Downs</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Implicit behavioral cloning</article-title>. <source>Proc. Mach. Learn. Res.</source> <volume>164</volume>, <fpage>158</fpage>&#x2013;<lpage>168</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v164/florence22a">https://proceedings.mlr.press/v164/florence22a</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Frans</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hafner</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2025</year>). &#x201c;<article-title>One step diffusion via shortcut models</article-title>,&#x201d; in <conf-name>The Thirteenth International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Freiberg</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Qualmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vien</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Neumann</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Diffusion for multi-embodiment grasping</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>10</volume>, <fpage>2694</fpage>&#x2013;<lpage>2701</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2025.3534065</pub-id>
</mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Fu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nachum</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Tucker</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <source>D4RL: datasets for deep data-driven reinforcement learning</source>. <comment>arXiv preprint arXiv:2004.07219</comment>.</mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Geng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Diffusion policies as multi-agent reinforcement learning strategies</article-title>,&#x201d; in <source>Lecture notes in computer science</source>, <fpage>356</fpage>&#x2013;<lpage>364</lpage>.</mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gervet</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Xian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gkanatsios</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Fragkiadaki</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Act3D: 3D feature field transformers for multi-task robotic manipulation</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>3949</fpage>&#x2013;<lpage>3965</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/gervet23a.html">https://proceedings.mlr.press/v229/gervet23a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gilles</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>E. Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Furmans</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wong</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Metagraspnetv2: all-in-one dataset enabling fast and reliable robotic bin picking via object relationship reasoning and dexterous grasping</article-title>. <source>IEEE Trans. Automation Sci. Eng.</source> <volume>21</volume>, <fpage>2302</fpage>&#x2013;<lpage>2320</lpage>. <pub-id pub-id-type="doi">10.1109/tase.2023.3328964</pub-id>
</mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gilles</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Furmans</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Rayyes</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>MetaMVUC: active learning for sample-efficient sim-to-real domain adaptation in robotic grasping</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>10</volume>, <fpage>3644</fpage>&#x2013;<lpage>3651</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2025.3544083</pub-id>
</mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Goyal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Blukis</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Chao</surname>
<given-names>Y.-W.</given-names>
</name>
<name>
<surname>Fox</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Rvt: robotic view transformer for 3d object manipulation</article-title>,&#x201d; in <conf-name>Conference on Robot Learning</conf-name>, <fpage>694</fpage>&#x2013;<lpage>710</lpage>.</mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Gu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Bao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Vector quantized diffusion model for text-to-image synthesis</article-title>,&#x201d; in <conf-name>2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <fpage>10686</fpage>&#x2013;<lpage>10696</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01043</pub-id>
</mixed-citation>
</ref>
<ref id="B42">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Lynch</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hausman</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Relay policy learning: solving long-horizon tasks via imitation and reinforcement learning</article-title>. <source>Proc. Mach. Learn. Res.</source>, <fpage>1025</fpage>&#x2013;<lpage>1037</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v100/gupta20a">https://proceedings.mlr.press/v100/gupta20a</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B43">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ha</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Florence</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Scaling up and distilling Down: language-guided robot skill acquisition</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>3766</fpage>&#x2013;<lpage>3777</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/ha23a.html">https://proceedings.mlr.press/v229/ha23a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B44">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ermon</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Generative adversarial imitation learning</article-title>,&#x201d; in <source>Advances in neural information processing systems</source>.</mixed-citation>
</ref>
<ref id="B45">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jain</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Denoising diffusion probabilistic models</article-title>,&#x201d; in <conf-name>Proceedings of the 34th International Conference on Neural Information Processing Systems</conf-name>.</mixed-citation>
</ref>
<ref id="B46">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ho</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Research</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Salimans</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Classifier-free diffusion guidance</article-title>,&#x201d; in <conf-name>NeurIPS 2021 Workshop on Deep Generative Models and Downstream Applications</conf-name>.</mixed-citation>
</ref>
<ref id="B47">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2025a</year>). <source>HGDiffuser: efficient task-oriented grasp generation via human-guided grasp diffusion models</source>. <comment>arXiv preprint arXiv:2503.00508</comment>.</mixed-citation>
</ref>
<ref id="B48">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Walters</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Platt</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Edge grasp network: a graph-based se (3)-invariant approach to grasp detection</article-title>,&#x201d; in <conf-name>2023 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>3882</fpage>&#x2013;<lpage>3888</lpage>. <pub-id pub-id-type="doi">10.1109/icra48891.2023.10160728</pub-id>
</mixed-citation>
</ref>
<ref id="B49">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yong</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Linghu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). &#x201c;<article-title>An embodied generalist agent in 3D world</article-title>,&#x201d; in <conf-name>Proceedings of the 41st International Conference on Machine Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B50">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Ze</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Institute</surname>
<given-names>Z.</given-names>
</name>
</person-group> (<year>2025b</year>). &#x201c;<article-title>Diffusion reward: learning rewards via conditional video diffusion</article-title>,&#x201d; in <conf-name>Computer Vision &#x2013; ECCV</conf-name>, <fpage>478</fpage>&#x2013;<lpage>495</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-72946-1_27</pub-id>
</mixed-citation>
</ref>
<ref id="B51">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Huang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Berenson</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2024b</year>). &#x201c;<article-title>Subgoal diffuser: coarse-to-fine subgoal generation to guide model predictive control for robot manipulation</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>16489</fpage>&#x2013;<lpage>16495</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10610189</pub-id>
</mixed-citation>
</ref>
<ref id="B52">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Iioka</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yoshida</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wada</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hatanaka</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Sugiura</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Multimodal diffusion segmentation model for object segmentation from manipulation instructions</article-title>,&#x201d; in <conf-name>IEEE International Conference on Intelligent Robots and Systems</conf-name>, <fpage>7590</fpage>&#x2013;<lpage>7597</lpage>. <pub-id pub-id-type="doi">10.1109/IROS55552.2023.10341402</pub-id>
</mixed-citation>
</ref>
<ref id="B53">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ikeda</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zakharov</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ko</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Irshad</surname>
<given-names>M. Z.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>Diffusionnocs: managing symmetry and uncertainty in sim2real multi-modal category-level pose estimation</article-title>,&#x201d; in <conf-name>2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>, <fpage>7406</fpage>&#x2013;<lpage>7413</lpage>. <pub-id pub-id-type="doi">10.1109/IROS58592.2024.10802487</pub-id>
</mixed-citation>
</ref>
<ref id="B54">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>James</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Arrojo</surname>
<given-names>D. R.</given-names>
</name>
<name>
<surname>Davison</surname>
<given-names>A. J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>RLBench: the robot learning benchmark &#x26; learning environment</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>5</volume>, <fpage>3019</fpage>&#x2013;<lpage>3026</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2020.2974707</pub-id>
</mixed-citation>
</ref>
<ref id="B55">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Janner</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Planning with diffusion for flexible behavior synthesis</article-title>. <source>Proc. 39th Int. Conf. Mach. Learn.</source> <volume>162</volume>, <fpage>9902</fpage>&#x2013;<lpage>9915</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v162/janner22a.html">https://proceedings.mlr.press/v162/janner22a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B56">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Svetlik</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Synergies between affordance and geometry: 6-dof grasp detection via implicit representations</source>. <comment>arXiv preprint arXiv:2104.01542</comment>.</mixed-citation>
</ref>
<ref id="B57">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Jolicoeur-Martineau</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Pich&#xe9;-Taillefer</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kachman</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Gotta Go fast with score-based generative models</article-title>,&#x201d; in <conf-name>The Symposium of Deep Learning and Differential Equations</conf-name>.</mixed-citation>
</ref>
<ref id="B58">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Kang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Efficient diffusion policies for offline reinforcement learning</article-title>,&#x201d; in <source>Advances in neural information processing systems</source>, <fpage>67195</fpage>&#x2013;<lpage>67212</lpage>.</mixed-citation>
</ref>
<ref id="B59">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kapelyukh</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Alzugaray</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Johns</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Dream2Real: zero-shot 3D object rearrangement with vision-language models</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>4796</fpage>&#x2013;<lpage>4803</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10611220</pub-id>
</mixed-citation>
</ref>
<ref id="B60">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kapelyukh</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Vosylius</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Johns</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>DALL-E-Bot: introducing web-scale diffusion models to robotics</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>8</volume>, <fpage>3956</fpage>&#x2013;<lpage>3963</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2023.3272516</pub-id>
</mixed-citation>
</ref>
<ref id="B61">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karras</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Aittala</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Aila</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Laine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Elucidating the design space of diffusion-based generative models</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>35</volume>, <fpage>26565</fpage>&#x2013;<lpage>26577</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2022">https://proceedings.neurips.cc/paper_files/paper/2022</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B62">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kasahara</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Engin</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chavan-Dafle</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Isler</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>RIC: rotate-inpaint-complete for generalizable scene reconstruction</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>2713</fpage>&#x2013;<lpage>2720</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10611694</pub-id>
</mixed-citation>
</ref>
<ref id="B63">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Katara</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Xian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Fragkiadaki</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Gen2Sim: scaling up robot learning in simulation with generative models</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>6672</fpage>&#x2013;<lpage>6679</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10610566</pub-id>
</mixed-citation>
</ref>
<ref id="B64">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ke</surname>
<given-names>T.-W.</given-names>
</name>
<name>
<surname>Gkanatsios</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Fragkiadaki</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>3D diffuser actor: policy diffusion with 3D scene representations</article-title>,&#x201d; in <conf-name>8th Annual Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B65">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Pertsch</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Karamcheti</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Balakrishna</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). &#x201c;<article-title>OpenVLA: an open-source vision-language-action model</article-title>,&#x201d; in <conf-name>8th Annual Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B66">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Matsunaga</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>K.-E.</given-names>
</name>
</person-group> (<year>2024b</year>). <article-title>Stitching sub-trajectories with conditional diffusion model for goal-conditioned offline RL</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>38</volume>, <fpage>13160</fpage>&#x2013;<lpage>13167</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v38i12.29215</pub-id>
</mixed-citation>
</ref>
<ref id="B67">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname>
<given-names>W. K.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Woo</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2024c</year>). <article-title>Robust policy learning via offline skill diffusion</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>38</volume>, <fpage>13177</fpage>&#x2013;<lpage>13184</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v38i12.29217</pub-id>
</mixed-citation>
</ref>
<ref id="B68">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Kirillov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Mintun</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ravi</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Rolland</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Gustafson</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). &#x201c;<article-title>Segment anything</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>, <fpage>4015</fpage>&#x2013;<lpage>4026</lpage>.</mixed-citation>
</ref>
<ref id="B69">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ko</surname>
<given-names>P.-C.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>S.-H.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J. B.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Learning to act from actionless videos through dense correspondences</article-title>,&#x201d; in <conf-name>The Twelth Internactional Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B70">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Krichen</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Generative adversarial networks</article-title>,&#x201d; in <conf-name>2023 14th International Conference on Computing Communication and Networking Technologies (ICCCNT)</conf-name>, <fpage>1</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1109/ICCCNT56998.2023.10306417</pub-id>
</mixed-citation>
</ref>
<ref id="B71">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tucker</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Fu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <source>Offline reinforcement learning: tutorial, review, and perspectives on open problems</source>. <comment>CoRR abs/2005.01643</comment>.</mixed-citation>
</ref>
<ref id="B72">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Knoll</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2025</year>). <source>Language-guided object-centric diffusion policy for generalizable and collision-aware robotic manipulation</source>. <comment>arXiv preprint arXiv:2407.00451</comment>.</mixed-citation>
</ref>
<ref id="B73">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2024a</year>). &#x201c;<article-title>ClickDiff: click to induce semantic contact map for controllable grasp generation with diffusion models</article-title>,&#x201d; in <conf-name>Proceedings of the 32nd ACM International Conference on Multimedia</conf-name>, <fpage>273</fpage>&#x2013;<lpage>281</lpage>. <pub-id pub-id-type="doi">10.1145/3664647.3680597</pub-id>
</mixed-citation>
</ref>
<ref id="B74">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Liao</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2024b</year>). <source>CogACT: a foundational vision-language-action model for synergizing cognition and action in robotic manipulation</source>. <comment>arXiv preprint arXiv:2411.19650</comment>.</mixed-citation>
</ref>
<ref id="B75">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zha</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Hierarchical diffusion for offline decision making</article-title>. <source>Proc. 40th Int. Conf. Mach. Learn.</source> <volume>202</volume>, <fpage>20035</fpage>&#x2013;<lpage>20064</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v202/li23ad.html">https://proceedings.mlr.press/v202/li23ad.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B76">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Belagali</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Shang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ryoo</surname>
<given-names>M. S.</given-names>
</name>
</person-group> (<year>2024c</year>). &#x201c;<article-title>Crossway diffusion: improving diffusion-based visuomotor policy via self-supervised learning</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>16841</fpage>&#x2013;<lpage>16849</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10610175</pub-id>
</mixed-citation>
</ref>
<ref id="B77">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Thickstun</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gulrajani</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Liang</surname>
<given-names>P. S.</given-names>
</name>
<name>
<surname>Hashimoto</surname>
<given-names>T. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Diffusion-LM improves controllable text generation</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>35</volume>, <fpage>4328</fpage>&#x2013;<lpage>4343</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2022">https://proceedings.neurips.cc/paper_files/paper/2022</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B78">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Shu</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2024d</year>). <source>ALDM-Grasping: diffusion-aided zero-shot sim-to-real transfer for robot grasping</source>. <comment>arXiv preprint arXiv:2403.11459</comment>.</mixed-citation>
</ref>
<ref id="B79">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ni</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Tomizuka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>AdaptDiffuser: diffusion models as adaptive self-evolving planners</article-title>. <source>Proc. 40th Int. Conf. Mach. Learn.</source> <volume>202</volume>, <fpage>20725</fpage>&#x2013;<lpage>20745</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v202/liang23e.html">https://proceedings.mlr.press/v202/liang23e.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B80">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Tomizuka</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>SkillDiffuser: interpretable hierarchical planning via skill abstractions in diffusion-based task execution</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <fpage>16467</fpage>&#x2013;<lpage>16476</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr52733.2024.01558</pub-id>
</mixed-citation>
</ref>
<ref id="B81">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lim</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Park</surname>
<given-names>F. C.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>EquiGraspFlow: SE (3)-Equivariant 6-DoF grasp pose generative flows</article-title>,&#x201d; in <conf-name>8th Annual Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B82">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lipman</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>R. T. Q.</given-names>
</name>
<name>
<surname>Ben-Hamu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nickel</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Flow matching for generative modeling</article-title>,&#x201d; in <conf-name>The Eleventh International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B83">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Feng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2023a</year>). <article-title>LIBERO: benchmarking knowledge transfer for lifelong robot learning</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>, <fpage>44776</fpage>&#x2013;<lpage>44791</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2023">https://proceedings.neurips.cc/paper_files/paper/2023</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B84">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <source>RDT-1B: a diffusion foundation model for bimanual manipulation</source>. <comment>arXiv preprint arXiv:2410.07864</comment>.</mixed-citation>
</ref>
<ref id="B85">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Ren</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). &#x201c;<article-title>Grounding dino: marrying dino with grounded pre-training for open-set object detection</article-title>,&#x201d; in <conf-name>Computer Vision &#x2013; ECCV 2024</conf-name>, <fpage>38</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-72970-6_3</pub-id>
</mixed-citation>
</ref>
<ref id="B86">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Hermans</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Chernova</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Paxton</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>StructDiffusion: language-guided creation of physically-valid structures using unseen objects</article-title>. <source>Robotics Sci. Syst</source>. <pub-id pub-id-type="doi">10.15607/RSS.2023.XIX.031</pub-id>
</mixed-citation>
</ref>
<ref id="B87">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Hermans</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Garg</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023c</year>). <article-title>Composable part-based manipulation</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>1300</fpage>&#x2013;<lpage>1315</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/liu23e.html">https://proceedings.mlr.press/v229/liu23e.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B88">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bao</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>DPM-Solver: a fast ODE solver for diffusion probabilistic model sampling in around 10 steps</article-title>,&#x201d; in <source>Advances in neural information processing systems</source>, <fpage>5775</fpage>&#x2013;<lpage>5787</lpage>.</mixed-citation>
</ref>
<ref id="B89">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). &#x201c;<article-title>Ugg: unified generative grasping</article-title>,&#x201d; in <conf-name>Computer Vision &#x2013; ECCV 2024</conf-name>, <fpage>414</fpage>&#x2013;<lpage>433</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-72855-6_24</pub-id>
</mixed-citation>
</ref>
<ref id="B90">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lucic</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kurach</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Google</surname>
<given-names>M. M.</given-names>
</name>
<name>
<surname>Bousquet</surname>
<given-names>B. O.</given-names>
</name>
<name>
<surname>Gelly</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Are GANs created equal? A large-scale study</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>30</volume>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2018">https://proceedings.neurips.cc/paper_files/paper/2018</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B91">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). <source>DexDiff: towards extrinsic dexterity manipulation of ungraspable objects in unrestricted environments</source>. <comment>arXiv preprint arXiv:2409.05493</comment>.</mixed-citation>
</ref>
<ref id="B92">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Patidar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Haughton</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>James</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024b</year>). &#x201c;<article-title>Hierarchical diffusion policy for kinematics-aware multi-task robotic manipulation</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <fpage>18081</fpage>&#x2013;<lpage>18090</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr52733.2024.01712</pub-id>
</mixed-citation>
</ref>
<ref id="B93">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Mandi</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Bharadhwaj</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Moens</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rajeswaran</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>CACTI: a framework for scalable multi-task multi-scene visual imitation learning</article-title>,&#x201d; in <conf-name>CoRL 2022 Workshop on Pre-training Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B94">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maria Scheikl</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gyenes</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Younis</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Haas</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Neumann</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wagner</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>LapGym-An open source framework for reinforcement learning in robot-assisted laparoscopic surgery</article-title>. <source>J. Mach. Learn. Res.</source> <volume>24</volume>, <fpage>1</fpage>&#x2013;<lpage>42</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://jmlr.org/papers/v24/23-0207.html">http://jmlr.org/papers/v24/23-0207.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B95">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Martinez</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Jacinto</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Montiel</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Rapidly exploring random trees for autonomous navigation in observable and uncertain environments</article-title>. <source>Int. J. Adv. Comput. Sci. Appl.</source> <volume>14</volume>. <pub-id pub-id-type="doi">10.14569/IJACSA.2023.0140399</pub-id>
</mixed-citation>
</ref>
<ref id="B96">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mattingley</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Boyd</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Receding horizon control</article-title>. <source>IEEE Control Syst. Mag.</source> <volume>31</volume>, <fpage>52</fpage>&#x2013;<lpage>65</lpage>. <pub-id pub-id-type="doi">10.1109/MCS.2011.940571</pub-id>
</mixed-citation>
</ref>
<ref id="B97">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mees</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Hermann</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Rosete-Beas</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Burgard</surname>
<given-names>W. B.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>CALVIN: a benchmark for language-conditioned policy learning for long-horizon robot manipulation tasks</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>7</volume>, <fpage>7327</fpage>&#x2013;<lpage>7334</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2022.3180108</pub-id>
</mixed-citation>
</ref>
<ref id="B98">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Meila</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Learning transferable visual models from natural language supervision</article-title>,&#x201d; in <conf-name>Proceedings of the 38th International Conference on Machine Learning</conf-name> (<publisher-name>PMLR</publisher-name>), <fpage>8748</fpage>&#x2013;<lpage>8763</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v139/radford21a">https://proceedings.mlr.press/v139/radford21a</ext-link>.</comment> </mixed-citation>
</ref>
<ref id="B99">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mendez-Mendez</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kaelbling</surname>
<given-names>L. P.</given-names>
</name>
<name>
<surname>Lozano-P&#xe9;rez</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Embodied lifelong learning for task and motion planning</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>2134</fpage>&#x2013;<lpage>2150</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/mendez-mendez23a.html">https://proceedings.mlr.press/v229/mendez-mendez23a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B100">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Meyer-Veit</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Rayyes</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gerstner</surname>
<given-names>A. O.</given-names>
</name>
<name>
<surname>Steil</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022a</year>). <source>Hyperspectral wavelength analysis with u-net for larynx cancer detection</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer Nature Switzerland</publisher-name>.</mixed-citation>
</ref>
<ref id="B101">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyer-Veit</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Rayyes</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gerstner</surname>
<given-names>A. O. H.</given-names>
</name>
<name>
<surname>Steil</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022b</year>). <article-title>Hyperspectral endoscopy using deep learning for laryngeal cancer segmentation</article-title>. <source>Artif. Neural Netw. Mach. Learn. &#x2013; ICANN</source> <volume>2022</volume>, <fpage>682</fpage>&#x2013;<lpage>694</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-15937-4_57</pub-id>
</mixed-citation>
</ref>
<ref id="B102">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Mildenhall</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Srinivasan</surname>
<given-names>P. P.</given-names>
</name>
<name>
<surname>Tancik</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Barron</surname>
<given-names>J. T.</given-names>
</name>
<name>
<surname>Ramamoorthi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ng</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>NeRF: representing scenes as neural radiance fields for view synthesis</article-title>,&#x201d; in <conf-name>Computer Vision &#x2013; ECCV 2020: 16th European Conference Proceedings, Part I</conf-name>, <conf-loc>Glasgow, UK</conf-loc>, <conf-date>August 23&#x2013;28, 2020</conf-date>, <fpage>405</fpage>&#x2013;<lpage>421</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-58452-8_24</pub-id>
</mixed-citation>
</ref>
<ref id="B103">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Mishra</surname>
<given-names>U. A.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>ReorientDiff: diffusion model based reorientation for object manipulation</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>10867</fpage>&#x2013;<lpage>10873</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10610749</pub-id>
</mixed-citation>
</ref>
<ref id="B104">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mishra</surname>
<given-names>U. A.</given-names>
</name>
<name>
<surname>Xue</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Generative skill chaining: long-horizon skill planning with diffusion models</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>2905</fpage>&#x2013;<lpage>2925</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/mishra23a.html">https://proceedings.mlr.press/v229/mishra23a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B105">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Misra</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Mish: a self regularized non-monotonic activation function</source>. <comment>arXiv preprint arXiv:1908.08681</comment>.</mixed-citation>
</ref>
<ref id="B106">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Mousavian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Eppner</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Fox</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>6-DOF GraspNet: variational grasp generation for object manipulation</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name>, <fpage>2901</fpage>&#x2013;<lpage>2910</lpage>. <pub-id pub-id-type="doi">10.1109/iccv.2019.00299</pub-id>
</mixed-citation>
</ref>
<ref id="B107">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Newbury</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chumbley</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Mousavian</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Eppner</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Leitner</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Deep learning approaches to grasp synthesis: a review</article-title>. <source>IEEE Trans. Robotics</source> <volume>39</volume>, <fpage>3994</fpage>&#x2013;<lpage>4015</lpage>. <pub-id pub-id-type="doi">10.1109/tro.2023.3280597</pub-id>
</mixed-citation>
</ref>
<ref id="B108">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>A. T.</given-names>
</name>
<name>
<surname>Pham</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Huber</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Vu</surname>
<given-names>M. N.</given-names>
</name>
</person-group> (<year>2025</year>). <source>FlowMP: learning motion fields for robot planning with conditional flow matching</source>. <comment>arXiv preprint arXiv:2503.06135</comment>.</mixed-citation>
</ref>
<ref id="B109">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Vu</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Vuong</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Vo</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). &#x201c;<article-title>Lightweight language-driven grasp detection using conditional consistency model</article-title>,&#x201d; in <conf-name>2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>, <fpage>13719</fpage>&#x2013;<lpage>13725doi</lpage>. <pub-id pub-id-type="doi">10.1109/IROS58592.2024.10802007</pub-id>
</mixed-citation>
</ref>
<ref id="B110">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Vu</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Van Vo</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Truong</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2024b</year>). &#x201c;<article-title>Language-conditioned affordance-pose detection in 3D point clouds</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>3071</fpage>&#x2013;<lpage>3078</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10610008</pub-id>
</mixed-citation>
</ref>
<ref id="B111">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nichol</surname>
<given-names>A. Q.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Improved denoising diffusion probabilistic models</article-title>. <source>Proc. 38th Int. Conf. Mach. Learn.</source> <volume>139</volume>, <fpage>8162</fpage>&#x2013;<lpage>8171</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v139/nichol21a.html">https://proceedings.mlr.press/v139/nichol21a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B112">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Oquab</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Darcet</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Moutakanni</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Vo</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Szafraniec</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khalidov</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <source>Dinov2: learning robust visual features without supervision</source>. <comment>arXiv preprint arXiv:2304.07193</comment>.</mixed-citation>
</ref>
<ref id="B113">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Junge</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hughes</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024a</year>). <source>Vision-language-action model and diffusion policy switching enables dexterous control of an anthropomorphic hand</source>. <comment>arXiv preprint arXiv:2410.14022</comment>.</mixed-citation>
</ref>
<ref id="B114">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Stachniss</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Popovi&#x107;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bennewitz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2024b</year>). &#x201c;<article-title>Exploiting priors from 3D diffusion models for RGB-based one-shot view planning</article-title>,&#x201d; in <conf-name>2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>, <fpage>13341</fpage>&#x2013;<lpage>13348</lpage>. <pub-id pub-id-type="doi">10.1109/IROS58592.2024.10802551</pub-id>
</mixed-citation>
</ref>
<ref id="B115">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Pan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jin</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Stachniss</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Popovi&#x107;</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Bennewitz</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2025</year>). <source>Dm-osvp&#x2b;&#x2b;: one-shot view planning using 3d diffusion models for active rgb-based object reconstruction</source>. <comment>arXiv preprint arXiv:2504.11674</comment>.</mixed-citation>
</ref>
<ref id="B116">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Pearce</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Rashid</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Kanervisto</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bignell</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Georgescu</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). &#x201c;<article-title>Imitating human behaviour with diffusion models</article-title>,&#x201d; in <conf-name>Deep Reinforcement Learning Workshop NeurIPS 2022</conf-name>. <pub-id pub-id-type="doi">10.48550/arXiv.2301.10677</pub-id>
</mixed-citation>
</ref>
<ref id="B117">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Peebles</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Scalable diffusion models with transformers</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE International Conference on Computer Vision</conf-name>, <fpage>4172</fpage>&#x2013;<lpage>4182</lpage>. <pub-id pub-id-type="doi">10.1109/ICCV51070.2023.00387</pub-id>
</mixed-citation>
</ref>
<ref id="B118">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Perez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Strub</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>De Vries</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Dumoulin</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Courville</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <source>FiLM: visual reasoning with a general conditioning layer</source>. <conf-name>Proceedings of the AAAI Conference on Artificial Intelligence</conf-name> <volume>32</volume>, (<issue>1</issue>). <pub-id pub-id-type="doi">10.1609/aaai.v32i1.11671</pub-id>
</mixed-citation>
</ref>
<ref id="B119">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Pertsch</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Stachowicz</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Ichter</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Driess</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Vuong</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <source>FAST: efficient action tokenization for vision-language-action models</source>. <comment>arXiv preprint arXiv:2501.0974</comment>.</mixed-citation>
</ref>
<ref id="B120">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Pfrommer</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Padmanabhan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ahn</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Umenberger</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Marcucci</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Mhammedi</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>On the sample complexity of imitation learning for smoothed model predictive control</article-title>,&#x201d; in <conf-name>2024 IEEE 63rd Conference on Decision and Control (CDC)</conf-name>, <fpage>1820</fpage>&#x2013;<lpage>1825doi</lpage>. <pub-id pub-id-type="doi">10.1109/CDC56724.2024.10886242</pub-id>
</mixed-citation>
</ref>
<ref id="B121">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Power</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Soltani-Zarrin</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Iba</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Berenson</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Sampling constrained trajectories using composable diffusion models</article-title>,&#x201d; in <conf-name>IROS 2023 Workshop on Differentiable Probabilistic Robotics: Emerging Perspectives on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B122">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Prasad</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Bohg</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Consistency policy accelerated visuomotor policies via consistency distillation</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.48550/arXiv.2405.07503</pub-id>
</mixed-citation>
</ref>
<ref id="B123">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Qi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Haramati</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Daniel</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Tamar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2025</year>). <source>Ec-diffuser: multi-object manipulation via entity-centric behavior generation</source>. <comment>arXiv prepping arXiv:2412.18907</comment>.</mixed-citation>
</ref>
<ref id="B124">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Qian</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Biza</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>ThinkGrasp: a vision-language system for strategic part grasping in clutter</article-title>,&#x201d; in <conf-name>8th Annual Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B125">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J. W.</given-names>
</name>
<name>
<surname>Hallacy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ramesh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Goh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Agarwal</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). &#x201c;<article-title>Learning transferable visual models from natural language supervision</article-title>,&#x201d; in <conf-name>International conference on machine learning</conf-name>, <fpage>8748</fpage>&#x2013;<lpage>8763</lpage>.</mixed-citation>
</ref>
<ref id="B126">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Rajeswaran</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Vezzani</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Todorov</surname>
<given-names>E.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <source>Learning complex dexterous manipulation with deep reinforcement learning and demonstrations</source>. <comment>arXiv preprint arXiv:1709.10087</comment>.</mixed-citation>
</ref>
<ref id="B127">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Ramesh</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Nichol</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Hierarchical text-conditional image generation with CLIP latents</source>. <comment>arXiv preprint arXiv:2204.06125</comment>.</mixed-citation>
</ref>
<ref id="B128">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ren</surname>
<given-names>A. Z.</given-names>
</name>
<name>
<surname>Lidard</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ankile</surname>
<given-names>L. L.</given-names>
</name>
<name>
<surname>Simeonov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Majumdar</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>Diffusion policy policy optimization</article-title>,&#x201d; in <conf-name>CoRL 2024 Workshop on Mastering Robot Manipulation in a World of Abundant Data</conf-name>.</mixed-citation>
</ref>
<ref id="B129">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reuss</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Erdin&#xe7;</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Gmurlu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wenzel</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lioutikov</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024a</year>). <article-title>Multimodal diffusion transformer: learning versatile behavior from multimodal goals</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.48550/arXiv.2407.05996</pub-id>
</mixed-citation>
</ref>
<ref id="B130">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reuss</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lioutikov</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Goal-conditioned imitation learning using score-based diffusion policies</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.48550/arXiv.2304.02532</pub-id>
</mixed-citation>
</ref>
<ref id="B131">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Reuss</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pari</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Agrawal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Lioutikov</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024b</year>). <source>Efficient diffusion transformer policies with mixture of expert denoisers for multitask learning</source>. <comment>arXiv preprint arXiv:2412.12953</comment>.</mixed-citation>
</ref>
<ref id="B132">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Rombach</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Blattmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lorenz</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Esser</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ommer</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022a</year>). &#x201c;<article-title>High-resolution image synthesis with latent diffusion models</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <fpage>10684</fpage>&#x2013;<lpage>10695</lpage>.</mixed-citation>
</ref>
<ref id="B133">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Rombach</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Blattmann</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lorenz</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Esser</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Ommer</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2022b</year>). &#x201c;<article-title>High-resolution image synthesis with latent diffusion models</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</conf-name>, <fpage>10684</fpage>&#x2013;<lpage>10695</lpage>.</mixed-citation>
</ref>
<ref id="B134">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>R&#xf6;mer</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>von Rohr</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schoellig</surname>
<given-names>A. P.</given-names>
</name>
</person-group> (<year>2024</year>). <source>Diffusion predictive control with constraints</source>. <comment>arXiv preprint/arXiv.2412.09342</comment>.</mixed-citation>
</ref>
<ref id="B135">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ronneberger</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Fischer</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Brox</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>U-Net: Convolutional networks for biomedical image segmentation</article-title>. <source>Med. Image Comput. Computer-Assisted Intervention &#x2013; MICCAI</source> <volume>2015</volume>, <fpage>234</fpage>&#x2013;<lpage>241</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-24574-4_28</pub-id>
</mixed-citation>
</ref>
<ref id="B136">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ross</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Bagnell</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Efficient reductions for imitation learning</article-title>. <source>Proc. Thirteen. Int. Conf. Artif. Intell. Statistics</source> <volume>9</volume>, <fpage>661</fpage>&#x2013;<lpage>668</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v9/ross10a.html">https://proceedings.mlr.press/v9/ross10a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B137">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Rouxel</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Ferrari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ivaldi</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mouret</surname>
<given-names>J.-B.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Flow matching imitation learning for multi-support manipulation</article-title>,&#x201d; in <conf-name>2024 IEEE-RAS 23rd International Conference on Humanoid Robots (Humanoids)</conf-name>, <fpage>528</fpage>&#x2013;<lpage>535</lpage>. <pub-id pub-id-type="doi">10.1109/Humanoids58906.2024.10769838</pub-id>
</mixed-citation>
</ref>
<ref id="B138">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ryu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Seo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>Diffusion-EDFs: Bi-equivariant denoising generative modeling on SE(3) for visual robotic manipulation</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name>, <fpage>18007</fpage>&#x2013;<lpage>18018</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr52733.2024.01705</pub-id>
</mixed-citation>
</ref>
<ref id="B139">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ryu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>J.-H.</given-names>
</name>
<name>
<surname>Choi</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Equivariant descriptor fields: SE(3)-equivariant energy-based models for end-to-end visual robotic manipulation learning</article-title>,&#x201d; in <conf-name>The Eleventh International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B140">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Saha</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mandadi</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Reddy</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Srikanth</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Agarwal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Sen</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>EDMP: ensemble-of-costs-guided diffusion for motion planning</article-title>,&#x201d; in <conf-name>2024 IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>10351</fpage>&#x2013;<lpage>10358</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA57147.2024.10610519</pub-id>
</mixed-citation>
</ref>
<ref id="B141">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Salimans</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Progressive distillation for fast sampling of diffusion models</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations (ICLR)</conf-name>.</mixed-citation>
</ref>
<ref id="B142">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Scheikl</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Schreiber</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Haas</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Freymuth</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Neumann</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Lioutikov</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Movement primitive diffusion: learning gentle robotic manipulation of deformable objects</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>9</volume>, <fpage>5338</fpage>&#x2013;<lpage>5345</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2024.3382529</pub-id>
</mixed-citation>
</ref>
<ref id="B143">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Seo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yoo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ryu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Lee</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <source>SE (3)-Equivariant robot learning and control: a tutorial survey</source>. <comment>arXiv preprint arXiv:2503.09829</comment>.</mixed-citation>
</ref>
<ref id="B144">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Shentu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Rajeswaran</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>From LLMs to actions: latent codes as bridges in hierarchical robot control</article-title>,&#x201d; in <conf-name>2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>, <fpage>8539</fpage>&#x2013;<lpage>8546</lpage>. <pub-id pub-id-type="doi">10.1109/IROS58592.2024.10801683</pub-id>
</mixed-citation>
</ref>
<ref id="B145">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>L. X.</given-names>
</name>
<name>
<surname>Sharma</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>T. Z.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Waypoint-based imitation learning for robotic manipulation</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>2195</fpage>&#x2013;<lpage>2209</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/shi23b.html">https://proceedings.mlr.press/v229/shi23b.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B146">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Welte</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gilles</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Rayyes</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). <source>vMF-Contact: uncertainty-aware evidential learning for probabilistic contact-grasp in noisy clutter</source>. <comment>arXiv preprint arXiv:2411.03591</comment>.</mixed-citation>
</ref>
<ref id="B147">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Shi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Welte</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Peng</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <source>VISO-Grasp: vision-language informed spatial object-centric 6-DoF active view planning and grasping in clutter and invisibility</source>. <comment>arXiv preprint arXiv:2503.12609</comment>.</mixed-citation>
</ref>
<ref id="B148">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Si</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Temel</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Kroemer</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Tilde: teleoperation for dexterous In-Hand manipulation learning with a DeltaHand</article-title>. <source>Robotics Sci. Syst</source>.</mixed-citation>
</ref>
<ref id="B149">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Simeonov</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Goyal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Manuelli</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Yen-Chen</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Sarmiento</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Rodriguez</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). &#x201c;<article-title>Shelving, stacking, hanging: relational pose diffusion for multi-modal rearrangement</article-title>,&#x201d; in <conf-name>Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B150">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Singh</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kalwar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Karim</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Sen</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Govindan</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Sridhar</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>Constrained 6-DoF grasp generation on complex shapes for improved dual-arm manipulation</article-title>,&#x201d; in <conf-name>2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>, <fpage>7344</fpage>&#x2013;<lpage>7350</lpage>. <pub-id pub-id-type="doi">10.1109/IROS58592.2024.10802268</pub-id>
</mixed-citation>
</ref>
<ref id="B151">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sohl-Dickstein</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Weiss</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Maheswaranathan</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ganguli</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep unsupervised learning using nonequilibrium thermodynamics</article-title>. <source>Proc. 32nd Int. Conf. Mach. Learn.</source> <volume>37</volume>, <fpage>2256</fpage>&#x2013;<lpage>2265</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v37/sohl-dickstein15.html">https://proceedings.mlr.press/v37/sohl-dickstein15.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B152">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Meng</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Ermon</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2021a</year>). &#x201c;<article-title>Denoising diffusion implicit models</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B153">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Detry</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Implicit grasp diffusion: bridging the gap between dense prediction and sampling-based grasping</article-title>,&#x201d; in <conf-name>8th Annual Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B154">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ermon</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Generative modeling by estimating gradients of the data distribution</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>32</volume>.</mixed-citation>
</ref>
<ref id="B155">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Song</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Sohl-Dickstein</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ermon</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Poole</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2021b</year>). &#x201c;<article-title>Score-based generative modeling through stochastic differential equations</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B156">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Suh</surname>
<given-names>H. T.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Dai</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tedrake</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Fighting uncertainty with gradients: offline reinforcement learning via diffusion score matching</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>2878</fpage>&#x2013;<lpage>2904</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/suh23a.html">https://proceedings.mlr.press/v229/suh23a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B157">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tarvainen</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Valpola</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Mean teachers are better role models: weight-averaged consistency targets improve semi-supervised deep learning results</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>30</volume>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2017">https://proceedings.neurips.cc/paper_files/paper/2017</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B158">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Team</surname>
<given-names>O. M.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Walke</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Pertsch</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Black</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Mees</surname>
<given-names>O.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <source>Octo: an open-source generalist robot policy</source>. <comment>arXiv preprint arXiv:2405.12213</comment>.</mixed-citation>
</ref>
<ref id="B159">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tobin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ray</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zaremba</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Domain randomization for transferring deep neural networks from simulation to the real world</article-title>,&#x201d; in <conf-name>2017 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>, <fpage>23</fpage>&#x2013;<lpage>30</lpage>. <pub-id pub-id-type="doi">10.1109/IROS.2017.8202133</pub-id>
</mixed-citation>
</ref>
<ref id="B160">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tremblay</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Prakash</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Acuna</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Brophy</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Jampani</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Anil</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). &#x201c;<article-title>Training deep networks with synthetic data: bridging the reality gap by domain randomization</article-title>,&#x201d; in <conf-name>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</conf-name>, <fpage>1082</fpage>&#x2013;<lpage>10828</lpage>. <pub-id pub-id-type="doi">10.1109/CVPRW.2018.00143</pub-id>
</mixed-citation>
</ref>
<ref id="B161">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tsagkas</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Rome</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ramamoorthy</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Aodha</surname>
<given-names>O. M.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>C. X.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>Click to grasp: zero-shot precise manipulation via visual diffusion descriptors</article-title>,&#x201d; in <conf-name>2024 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</conf-name>, <fpage>11610</fpage>&#x2013;<lpage>11617</lpage>. <pub-id pub-id-type="doi">10.1109/IROS58592.2024.10801488</pub-id>
</mixed-citation>
</ref>
<ref id="B162">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Urain</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Funk</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Peters</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chalvatzaki</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>SE(3)-DiffusionFields: learning smooth cost functions for joint grasp and motion optimization through diffusion</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE International Conference on Robotics and Automation (ICRA)</conf-name>, <fpage>5923</fpage>&#x2013;<lpage>5930</lpage>. <pub-id pub-id-type="doi">10.1109/ICRA48891.2023.10161569</pub-id>
</mixed-citation>
</ref>
<ref id="B163">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Venkatraman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Khaitan</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Akella</surname>
<given-names>R. T.</given-names>
</name>
<name>
<surname>Dolan</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Berseth</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Reasoning with latent diffusion in offline reinforcement learning</source>. <comment>arXiv preprint arXiv:2309.06599</comment>.</mixed-citation>
</ref>
<ref id="B164">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vosylius</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Seo</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Uru&#xe7;</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>James</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Render and diffuse: aligning image and action spaces for diffusion-based behaviour cloning</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.15607/RSS.2024.XX.051</pub-id>
</mixed-citation>
</ref>
<ref id="B165">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Vuong</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Vu</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Nguyen</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Le</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Vo</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>Language-driven grasp detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>, <fpage>17902</fpage>&#x2013;<lpage>17912</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr52733.2024.01695</pub-id>
</mixed-citation>
</ref>
<ref id="B166">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Shi</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Fei-Fei</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C. K.</given-names>
</name>
</person-group> (<year>2024a</year>). &#x201c;<article-title>DexCap: scalable and portable mocap data collection system for dexterous manipulation</article-title>,&#x201d; in <conf-name>2nd Workshop on Dexterous Manipulation: Design, Perception and Control (RSS)</conf-name>.</mixed-citation>
</ref>
<ref id="B167">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Adelson</surname>
<given-names>E. H.</given-names>
</name>
<name>
<surname>Tedrake</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2024b</year>). <article-title>PoCo: policy composition from and for heterogeneous robot learning</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.48550/arXiv.2402.02511</pub-id>
</mixed-citation>
</ref>
<ref id="B168">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.-K.</given-names>
</name>
<name>
<surname>Xing</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wei</surname>
<given-names>Y.-L.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>X.-M.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>W.-S.</given-names>
</name>
</person-group> (<year>2024c</year>). &#x201c;<article-title>Single-view scene point cloud human grasp generation</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>, <fpage>831</fpage>&#x2013;<lpage>841</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr52733.2024.00085</pub-id>
</mixed-citation>
</ref>
<ref id="B169">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Hunt</surname>
<given-names>J. J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2023a</year>). <source>Diffusion policies as an expressive policy class for offline reinforcement learning</source>. <comment>arXiv preprint arXiv:2208.06193</comment>.</mixed-citation>
</ref>
<ref id="B170">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Oba</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Yoneda</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Walter</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stadie</surname>
<given-names>B. C.</given-names>
</name>
</person-group> (<year>2023b</year>). <article-title>Cold diffusion on the replay buffer: learning to plan from known good States</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>3277</fpage>&#x2013;<lpage>3291</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/wang23e.html">https://proceedings.mlr.press/v229/wang23e.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B171">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Watson</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Chan</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ho</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Norouzi</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Learning fast samplers for diffusion models by differentiating trough sample quality</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations (ICLR)</conf-name>.</mixed-citation>
</ref>
<ref id="B172">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Welte</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rayyes</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2025</year>). <source>Interactive imitation learning for dexterous robotic manipulation: challenges and perspectives &#x2013; a survey</source>. <comment>arXiv prepint arXiv:2506.00098</comment>.</mixed-citation>
</ref>
<ref id="B173">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <source>Diffusion-VLA: scaling robot foundation models via unified diffusion and autoregression</source>. <comment>arXiv preprint arXiv:2412.03293</comment>.</mixed-citation>
</ref>
<ref id="B174">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>K.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>TinyVLA: toward fast, data-efficient vision-language-action models for robotic manipulation</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>10</volume>, <fpage>3988</fpage>&#x2013;<lpage>3995</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2025.3544909</pub-id>
</mixed-citation>
</ref>
<ref id="B175">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weng</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Kragic</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lundell</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>DexDiffuser: generating dexterous grasps with diffusion models</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>9</volume>, <fpage>11834</fpage>&#x2013;<lpage>11840</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2024.3498776</pub-id>
</mixed-citation>
</ref>
<ref id="B176">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Gan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). <source>Unidexfpm: universal dexterous functional pre-grasp manipulation via diffusion policy</source>. <comment>arXiv preprint arXiv:2403.12421</comment>.</mixed-citation>
</ref>
<ref id="B177">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Learning score-based grasping primitive for human-assisting dexterous grasping</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>, <fpage>22132</fpage>&#x2013;<lpage>22150</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2023">https://proceedings.neurips.cc/paper_files/paper/2023</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B178">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Gan</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Dong</surname>
<given-names>H.</given-names>
</name>
</person-group> (<year>2024b</year>). <article-title>Learning score-based grasping primitive for human-assisting dexterous grasping</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2024">https://proceedings.neurips.cc/paper_files/paper/2024</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B179">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Xian</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Gkanatsios</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gervet</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Ke</surname>
<given-names>T.-W.</given-names>
</name>
<name>
<surname>Fragkiadaki</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>ChainedDiffuser: unifying trajectory diffusion and keypose prediction for robotic manipulation</article-title>,&#x201d; in <conf-name>7th Annual Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B180">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Chi</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Veloso</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Song</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>XSkill: Cross embodiment skill discovery</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>3536</fpage>&#x2013;<lpage>3555</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/xu23a.html">https://proceedings.mlr.press/v229/xu23a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B181">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Xu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Loz&#xe1;no-P&#xe9;rez</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Pack Kaebling</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Hsu</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2024</year>). <source>&#x201c;set it up!&#x201d;: functional object arrangement with compositional generative models</source>. <comment>arXiv preprint arXiv:2405.11928</comment>.</mixed-citation>
</ref>
<ref id="B182">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kamyar</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ghasemipour</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Tompson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kaelbling</surname>
<given-names>L.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). &#x201c;<article-title>Learning interactive real-world simulators</article-title>,&#x201d; in <conf-name>The Twelfth International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B183">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Mao</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tenenbaum</surname>
<given-names>J. B.</given-names>
</name>
<name>
<surname>Lozano-P&#xe9;rez</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Compositional diffusion-based continuous constraint solvers</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>3242</fpage>&#x2013;<lpage>3265</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/yang23d.html">https://proceedings.mlr.press/v229/yang23d.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B184">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ye</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kitani</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Tulsiani</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). &#x201c;<article-title>G-HOP: generative hand-object prior for interaction reconstruction and grasp synthesis</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</conf-name>, <fpage>1911</fpage>&#x2013;<lpage>1920</lpage>. <pub-id pub-id-type="doi">10.1109/cvpr52733.2024.00187</pub-id>
</mixed-citation>
</ref>
<ref id="B185">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Xie</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Pang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Latent diffusion energy-based model for interpretable text modelling</article-title>. <source>Proc. 39th Int. Conf. Mach. Learn.</source> <volume>162</volume>, <fpage>25702</fpage>&#x2013;<lpage>25720</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v162/yu22h.html?ref=https://githubhelp.com">https://proceedings.mlr.press/v162/yu22h.html?ref&#x3d;https://githubhelp.com</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B186">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Quillen</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>He</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Julian</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hausman</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Finn</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Meta-world: a benchmark and evaluation for multi-task and Meta reinforcement learning</article-title>. <source>Proc. Conf. Robot. Learn.</source> <volume>100</volume>, <fpage>1094</fpage>&#x2013;<lpage>1100</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v100/yu20a">https://proceedings.mlr.press/v100/yu20a</ext-link>
</comment>
</mixed-citation>
</ref>
<ref id="B187">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Stone</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tompson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Brohan</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Scaling robot learning with semantic data augmentation through diffusion models</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.48550/arXiv.2211.04604</pub-id>
</mixed-citation>
</ref>
<ref id="B188">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zare</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kebria</surname>
<given-names>P. M.</given-names>
</name>
<name>
<surname>Khosravi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nahavandi</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A survey of imitation learning: Algorithms, recent developments, and challenges</article-title>. <source>IEEE Trans. Cybern.</source> <volume>54</volume>, <fpage>7173</fpage>&#x2013;<lpage>7186</lpage>. <pub-id pub-id-type="doi">10.1109/tcyb.2024.3395626</pub-id>
<pub-id pub-id-type="pmid">39024072</pub-id>
</mixed-citation>
</ref>
<ref id="B189">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ze</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Y.-H.</given-names>
</name>
<name>
<surname>Macaluso</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ge</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Ye</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>GNFactor: multi-task real robot learning with generalizable neural feature fields</article-title>. <source>Proc. 7th Conf. Robot Learn.</source> <volume>229</volume>, <fpage>284</fpage>&#x2013;<lpage>301</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.mlr.press/v229/ze23a.html">https://proceedings.mlr.press/v229/ze23a.html</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B190">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ze</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>3D diffusion policy: generalizable visuomotor policy learning via simple 3D representations</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.48550/arXiv.2403.0395</pub-id>
</mixed-citation>
</ref>
<ref id="B191">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zeng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>H.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>LVDiffusor: distilling functional rearrangement priors from large models into diffusor</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>9</volume>, <fpage>8258</fpage>&#x2013;<lpage>8265</lpage>. <pub-id pub-id-type="doi">10.1109/LRA.2024.3438036</pub-id>
</mixed-citation>
</ref>
<ref id="B192">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Lu</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2024a</year>). &#x201c;<article-title>Language control diffusion: efficiently scaling through space, time, and tasks</article-title>,&#x201d; in <conf-name>International Conference on Learning Representations</conf-name>.</mixed-citation>
</ref>
<ref id="B193">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Gienger</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2025</year>). <source>Affordance-based robot manipulation with flow matching</source>. <comment>arXiv preprint arXiv:2409.01083</comment>.</mixed-citation>
</ref>
<ref id="B194">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Geng</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024b</year>). &#x201c;<article-title>DexGraspNet 2.0: learning generative dexterous grasping in large-scale synthetic cluttered scenes</article-title>,&#x201d; in <conf-name>8th Annual Conference on Robot Learning</conf-name>.</mixed-citation>
</ref>
<ref id="B195">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Hong</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fang</surname>
<given-names>X.</given-names>
</name>
<etal/>
</person-group> (<year>2024c</year>). <source>NaVid: video-based VLM plans the next step for vision-and-language navigation</source>. <comment>arXiv preprint arXiv:2402.15852</comment>.</mixed-citation>
</ref>
<ref id="B196">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>An</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hu</surname>
<given-names>Z.</given-names>
</name>
<etal/>
</person-group> (<year>2024d</year>). <source>ManiDext: hand-object manipulation synthesis via continuous correspondence embeddings and residual-guided diffusion</source>. <comment>arXiv preprint arXiv:2409.09300</comment>.</mixed-citation>
</ref>
<ref id="B197">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>Q.</given-names>
</name>
<etal/>
</person-group> (<year>2024e</year>). <source>VLABench: a large-scale benchmark for language-conditioned robotics manipulation with long-horizon reasoning tasks</source>. <comment>arXiv preprint arXiv:2412.18194</comment>.</mixed-citation>
</ref>
<ref id="B198">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kumar</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Gupta</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2024f</year>). <article-title>Diffusion meets DAgger: supercharging eye-in-hand imitation learning</article-title>. <source>Robotics Sci. Syst.</source> <pub-id pub-id-type="doi">10.48550/arXiv.2402.17768</pub-id>
</mixed-citation>
</ref>
<ref id="B199">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhai</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Susskind</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jaitly</surname>
<given-names>N.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>PLANNER: generating diversified paragraph via latent language diffusion model</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>, <fpage>80178</fpage>&#x2013;<lpage>80190</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2023">https://proceedings.neurips.cc/paper_files/paper/2023</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B200">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Cheng</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yao</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>H. J.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Nl2contact: natural language guided 3d hand-object contact modeling with diffusion model</article-title>. <source>Comput. Vis. &#x2013; ECCV</source> <volume>2024</volume>, <fpage>284</fpage>&#x2013;<lpage>300</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-031-73390-1_17</pub-id>
</mixed-citation>
</ref>
<ref id="B201">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zheng</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Jang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2024g</year>). <source>GRAPE: generalizing robot policy via preference alignment</source>. <comment>arXiv preprint arXiv:2411.19309</comment>.</mixed-citation>
</ref>
<ref id="B202">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Yuan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2024h</year>). <source>DexGrasp-Diffusion: diffusion-based unified functional grasp synthesis method for multi-dexterous robotic hands</source>. <comment>arXiv preprint arXiv:2407.09899</comment>.</mixed-citation>
</ref>
<ref id="B203">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Bogdanovic</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Luo</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Tohme</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Darvish</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Aspuru-Guzik</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <source>AnyPlace: learning generalized object placement for robot manipulation</source>. <comment>arXiv preprint arXiv:2502.04531</comment>.</mixed-citation>
</ref>
<ref id="B204">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Qiu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yan</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>3D-VLA: a 3D vision-language-action generative world model</article-title>. <source>Proc. 41st Int. Conf. Mach. Learn.</source> <volume>235</volume>, <fpage>61229</fpage>&#x2013;<lpage>61245</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://dl.acm.org/doi/abs/10.5555/3692070.3694603">https://dl.acm.org/doi/abs/10.5555/3692070.3694603</ext-link>.</comment>
</mixed-citation>
</ref>
<ref id="B205">
<mixed-citation publication-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Allen-Blanchette</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2025</year>). <source>GAGrasp: geometric algebra diffusion for dexterous grasping</source>. <comment>arXiv preprint arXiv:2503.04123</comment>.</mixed-citation>
</ref>
<ref id="B206">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Blessing</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Celik</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Jia</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Neumann</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2024a</year>). <article-title>Variational distillation of diffusion policies into mixture of experts</article-title>. in <conf-name>The thirty-eighth annual conference on neural information processing systems</conf-name>.</mixed-citation>
</ref>
<ref id="B207">
<mixed-citation publication-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Yeung</surname>
<given-names>D.-Y.</given-names>
</name>
<name>
<surname>Gan</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2024b</year>). <article-title>RoboDreamer: learning compositional world models for robot imagination</article-title>. in <conf-name>Forty-first international conference on machine learning</conf-name>.</mixed-citation>
</ref>
<ref id="B208">
<mixed-citation publication-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Du</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Xu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Shen</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Xiao</surname>
<given-names>W.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Adaptive online replanning with diffusion models</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>36</volume>, <fpage>44000</fpage>&#x2013;<lpage>44016</lpage>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://proceedings.neurips.cc/paper_files/paper/2023">https://proceedings.neurips.cc/paper_files/paper/2023</ext-link>.</comment>
</mixed-citation>
</ref>
</ref-list>
</back>
</article>