<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Robot. AI</journal-id>
<journal-title>Frontiers in Robotics and AI</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Robot. AI</abbrev-journal-title>
<issn pub-type="epub">2296-9144</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1537470</article-id>
<article-id pub-id-type="doi">10.3389/frobt.2025.1537470</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Robotics and AI</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Learning to suppress tremors: a deep reinforcement learning-enabled soft exoskeleton for Parkinson&#x2019;s patients</article-title>
<alt-title alt-title-type="left-running-head">Endrei et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/frobt.2025.1537470">10.3389/frobt.2025.1537470</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Endrei</surname>
<given-names>Tam&#xe1;s</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2890974/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>F&#xf6;ldi</surname>
<given-names>S&#xe1;ndor</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3043991/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Makk</surname>
<given-names>&#xc1;d&#xe1;m</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Cserey</surname>
<given-names>Gy&#xf6;rgy</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/4242/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Faculty of Information Technology and Bionics</institution>, <institution>P&#xe1;zm&#xe1;ny P&#xe9;ter Catholic University</institution>, <addr-line>Budapest</addr-line>, <country>Hungary</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Jedlik Innovation Ltd.</institution>, <addr-line>Budapest</addr-line>, <country>Hungary</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Andr&#xe1;s Pet&#x151; Faculty</institution>, <institution>Semmelweis University</institution>, <addr-line>Budapest</addr-line>, <country>Hungary</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/659730/overview">Alessandro Filippeschi</ext-link>, Sant&#x2019;Anna School of Advanced Studies, Italy</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/514541/overview">Ali Foroutannia</ext-link>, University of Canberra, Australia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1560264/overview">Yali Liu</ext-link>, Beijing Institute of Technology, China</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Tam&#xe1;s Endrei, <email>endrei.tamas@itk.ppke.hu</email>Gy&#xF6;rgy Cserey, <email>cserey.gyorgy@itk.ppke.hu</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>05</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1537470</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>11</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>03</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Endrei, F&#xf6;ldi, Makk and Cserey.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Endrei, F&#xf6;ldi, Makk and Cserey</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Neurological tremors, prevalent among a large population, are one of the most rampant movement disorders. Biomechanical loading and exoskeletons show promise in enhancing patient well-being, but traditional control algorithms limit their efficacy in dynamic movements and personalized interventions. Furthermore, a pressing need exists for more comprehensive and robust validation methods to ensure the effectiveness and generalizability of proposed solutions.</p>
</sec>
<sec>
<title>Methods</title>
<p>This paper proposes a physical simulation approach modeling multiple arm joints and tremor propagation. This study also introduces a novel adaptable reinforcement learning environment tailored for disorders with tremors. We present a deep reinforcement learning-based encoder-actor controller for Parkinson&#x2019;s tremors in various shoulder and elbow joint axes displayed in dynamic movements.</p>
</sec>
<sec>
<title>Results</title>
<p>Our findings suggest that such a control strategy offers a viable solution for tremor suppression in real-world scenarios.</p>
</sec>
<sec>
<title>Discussion</title>
<p>By overcoming the limitations of traditional control algorithms, this work takes a new step in adapting biomechanical loading into the everyday life of patients. This work also opens avenues for more adaptive and personalized interventions in managing movement disorders.</p>
</sec>
</abstract>
<kwd-group>
<kwd>deep reinforcement learning</kwd>
<kwd>soft exoskeleton</kwd>
<kwd>Parkinson&#x2019;s disease</kwd>
<kwd>tremor</kwd>
<kwd>physics simulation</kwd>
<kwd>human&#x2013;robot interaction</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Biomedical Robotics</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>Neurodegenerative diseases are characterized by the loss of neurons in the central nervous system, which can impact an individual&#x2019;s quality of life by causing cognitive, motor, or behavioral symptoms <xref ref-type="bibr" rid="B33">Lamptey et al. (2022)</xref>. The occurrence of these disorders is expected to increase, partly due to the recent growth in the aging population <xref ref-type="bibr" rid="B25">Heemels (2016)</xref>. Neurological tremors are the most common of the movement disorders <xref ref-type="bibr" rid="B37">Louis et al. (1995)</xref>, present in multiple neurodegenerative disorders such as essential tremor <xref ref-type="bibr" rid="B10">Deuschl et al. (1998)</xref> and Parkinson&#x2019;s disease <xref ref-type="bibr" rid="B35">Lang and Lozano (1998b)</xref>, <xref ref-type="bibr" rid="B34">Lang and Lozano (1998a)</xref>. Tremors can be described as involuntary, oscillating, or rhythmic movements <xref ref-type="bibr" rid="B4">Bhatia et al. (2018)</xref>. Although these are not life-threatening, movement disorders pose serious difficulties in daily activities, functional disabilities, and social inconvenience, as well as difficulties performing tasks that require fine motor skills for two-thirds of the affected patients <xref ref-type="bibr" rid="B50">Rocon et al. (2007b)</xref>.</p>
<p>Although there is no cure for neurodegenerative diseases, current treatments aim to alleviate symptoms and enhance patient well-being. Invasive options, such as deep brain stimulation <xref ref-type="bibr" rid="B42">Oliveira et al. (2023)</xref>; <xref ref-type="bibr" rid="B15">Faraji et al. (2023)</xref>, neurosurgery <xref ref-type="bibr" rid="B1">Albano et al. (2023)</xref>, and stem cell therapies <xref ref-type="bibr" rid="B26">Heris et al. (2022)</xref>, can be effective but often come with high costs and severe side effects. Non-invasive treatments have been explored, ranging from medication <xref ref-type="bibr" rid="B2">Asil et al. (2020)</xref> and traditional Chinese therapies <xref ref-type="bibr" rid="B6">Cai et al. (2023)</xref> to advanced wearable technologies. These include robotic exoskeletons <xref ref-type="bibr" rid="B49">Rocon et al. (2007a)</xref>; <xref ref-type="bibr" rid="B27">Herrnstadt and Menon (2016)</xref> and soft exoskeletons <xref ref-type="bibr" rid="B55">Skaramagkas et al. (2020)</xref>; <xref ref-type="bibr" rid="B3">Awantha et al. (2020)</xref>; <xref ref-type="bibr" rid="B65">Zahedi et al. (2021)</xref>, as well as functional electrical stimulation (FES) devices <xref ref-type="bibr" rid="B13">Dosen et al. (2014)</xref>; <xref ref-type="bibr" rid="B29">Jitkritsadakul et al. (2017)</xref>, which use electrical stimulation. Additionally, afferent neuroprostheses have been developed to stimulate the patient&#x2019;s central nervous system <xref ref-type="bibr" rid="B44">Pascual-Valdunciel et al. (2020)</xref>; <xref ref-type="bibr" rid="B11">Dideriksen et al. (2017)</xref>. Of all the non-invasive treatment options, the use of exoskeletons has been proven to be the most efficient method for the suppression of tremors <xref ref-type="bibr" rid="B36">Lora-Millan et al. (2021)</xref>.</p>
<p>Wearable exoskeleton research mainly focused on reducing the weight of exoskeletons <xref ref-type="bibr" rid="B64">Yi et al. (2019)</xref>; <xref ref-type="bibr" rid="B61">Wang et al. (2023)</xref> due to their bulk and weight limiting their adoption. Therefore, control algorithms were not the main interest of these studies, which often utilized repetitive control <xref ref-type="bibr" rid="B49">Rocon et al. (2007a)</xref>, traditional control methods <xref ref-type="bibr" rid="B27">Herrnstadt and Menon (2016)</xref>; <xref ref-type="bibr" rid="B66">Zhou et al. (2017)</xref>; <xref ref-type="bibr" rid="B64">Yi et al. (2019)</xref>; <xref ref-type="bibr" rid="B65">Zahedi et al. (2021)</xref>, tremor frequency noise filtering <xref ref-type="bibr" rid="B58">Taheri et al. (2013)</xref>, or equivalent-input-disturbance (EID) tremor suppression <xref ref-type="bibr" rid="B63">Xie et al. (2024)</xref>. Traditional control methods, though widely used, have significant limitations. They are typically validated on low-degree-of-freedom systems and under static conditions, overlooking tremor propagation and the natural frequencies of voluntary movements. Evaluations often rely on healthy subjects mimicking tremors, which fails to capture the multi-harmonic characteristics of Parkinson&#x2019;s tremors. Additionally, these methods do not quantify or account for interference with voluntary motion. For dynamic movements, traditional methods require either time-consuming patient-specific training with human-in-the-loop optimization <xref ref-type="bibr" rid="B54">Siviy et al. (2023)</xref>; <xref ref-type="bibr" rid="B12">Ding et al. (2018)</xref> or manual rule design for each activity, limiting the scalability and adoption of wearable robotics <xref ref-type="bibr" rid="B56">Slade et al. (2022)</xref>. In contrast, recent advances in deep reinforcement learning (DRL) have shown promise in managing stochastic action spaces in robotics <xref ref-type="bibr" rid="B28">Jin et al. (2022)</xref>; <xref ref-type="bibr" rid="B31">Kaufmann et al. (2023)</xref>; <xref ref-type="bibr" rid="B23">Haarnoja et al. (2023)</xref> and are gaining traction for rehabilitation exoskeletons, as DRL enables simulation-based training without additional patient involvement <xref ref-type="bibr" rid="B39">Luo et al. (2021)</xref>, <xref ref-type="bibr" rid="B38">Luo et al. (2023)</xref>, <xref ref-type="bibr" rid="B40">Luo et al. (2024)</xref>.</p>
<p>Therefore, our work aims to incorporate recent advances in DRL and makes the following central contributions to the field of biomechanical loading exoskeletons:<list list-type="simple">
<list-item>
<p>&#x2022; We create a human&#x2013;exoskeleton simulation environment that is capable of simulating multiple different dynamic movements, different types of tremors, and human&#x2013;exoskeleton interactions.</p>
</list-item>
<list-item>
<p>&#x2022; We propose a model-free deep RL-based tremor-suppression controller capable of suppressing generated tremors across various axes of the shoulder and elbow joints during a multitude of dynamic movements.</p>
</list-item>
<list-item>
<p>&#x2022; We demonstrate that the soft exoskeleton <xref ref-type="fig" rid="F1">Figure 1A</xref>, coupled with our DRL-based controller, can accurately mitigate the effect of generated tremors.</p>
</list-item>
</list>
</p>
<p>The result is an intelligent tremor-suppression controller that minimizes its effects on the patient&#x2019;s original movements and posture, with no additional training required from the patient to adapt to the exoskeleton.</p>
<p>In the following sections, we detail the underlying physical simulation and the DRL framework, describe the experimental setup and evaluation metrics, present our results, and discuss the implications of our approach for future wearable robotics in the treatment of neurodegenerative movement disorders.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<sec id="s2-1">
<title>2.1 Tremor-suppression physical simulation</title>
<p>To facilitate the training of a reinforcement learning-based controller, we established a physical simulation environment to ensure a secure and cost-effective learning process. The simulation uses a human torso model with the addition of the right arm, in which the tremors induced will be suppressed. The simulation environment uses the Pybullet physics engine <xref ref-type="bibr" rid="B8">Coumans and Bai (2016)</xref> and the Open AI gym <xref ref-type="bibr" rid="B5">Brockman et al. (2016)</xref> to create a reinforcement learning environment.</p>
<p>The human&#x2013;exoskeleton simulation is made up of three distinct parts. The movements were recorded using two Velcro sleeves fixed around the upper and lower arm, with an additional inertial measurement unit (IMU) sensor fixed on the scapula of the right arm, as presented in <xref ref-type="fig" rid="F1">Figure 1B</xref>. These parts of the simulation are illustrated in <xref ref-type="fig" rid="F1">Figure 2</xref>, and described in the following sections.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>The anatomy of the tremor-suppression exoskeleton and the corresponding inertial measurement unit (IMU) reference movement acquisition system. <bold>(A)</bold> The soft-robotic exoskeleton used in the tremor-suppression simulations. <bold>(B)</bold> The IMU reference movement acquisition system. <bold>(A)</bold> Actuator positions. <bold>(B)</bold> IMU sensor positions.</p>
</caption>
<graphic xlink:href="frobt-12-1537470-g001.tif"/>
</fig>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>The parts of the simulation. Voluntary movements represent the trajectory of the movement in which involuntary movement tremors are generated, which the exoskeleton tries to suppress using its actuators.</p>
</caption>
<graphic xlink:href="frobt-12-1537470-g002.tif"/>
</fig>
<sec id="s2-1-1">
<title>2.1.1 Acquiring reference movements</title>
<p>The reference movements represent the patients&#x2019; voluntary movements, which act as the base trajectory for the environments. In the reference motions, we have recorded four distinct movements: shoulder flexion/extension, shoulder abduction/adduction, elbow flexion/extension, and the external rotation of the shoulder. Two distinct recordings are used for each distinct movement pattern for training to add variability and improve the robustness of the controller.</p>
<p>Of the five IMU sensors this system possessed, IMU 2 and IMU 4 were chosen. From the accelerations and angular accelerations measured, we could approximate the quaternions of the shoulder and elbow joints using an extended Kalman filter <xref ref-type="bibr" rid="B62">Welch and Bishop (1995)</xref>. Finally, the quaternions were transformed into Euler angles, which were used in the simulation.</p>
<p>Verbal informed consent was exchanged between the authors and the subject when planning, preparing, and executing the IMU measurements.</p>
</sec>
<sec id="s2-1-2">
<title>2.1.2 Generating tremors</title>
<p>Our generated involuntary movements can be described by three attributes: their amplitude, frequency, and the time duration during which the tremor effects are present. The third attribute can be disregarded to ensure a more computationally effective simulation and training of the control. Therefore, tremorous movement parts are present at every simulation time step.</p>
<p>In our simulations, we utilized Parkinson&#x2019;s disease tremors due to their well-documented and well-understood characteristics. Tremors present in Parkinson&#x2019;s disease can be described as a second-order non-linear stochastic process <xref ref-type="bibr" rid="B58">Taheri et al. (2013)</xref>, which can be approximated by the superposition of sine waves <xref ref-type="bibr" rid="B48">Riviere et al. (1997)</xref>.</p>
<p>In this paper, we approximated these tremors by two sine waves with given parameters based on <xref ref-type="bibr" rid="B58">Taheri et al. (2013)</xref>. This way, the approximation contains <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mn>96.4</mml:mn>
<mml:mo>&#xb1;</mml:mo>
<mml:mn>1.39</mml:mn>
<mml:mi>%</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> of the original energy of the tremor.</p>
<p>From the two main frequency ranges, we randomly sampled values for both harmonics independent from each other and added Gaussian noise to their sum, which was then consequentially normalized. To ensure a wide range of possible tremor cases, a vector containing the given arm joint axis was used to specify which joint axis was affected by tremors.</p>
<p>Finally, from this created vector, which contains the tremor acceleration values for each joint axis in the arm, we transformed these values into torque values based on the measurements done by <xref ref-type="bibr" rid="B32">Ketteringham et al. (2014)</xref>.</p>
<p>In tremor instances, where the effect of tremor impacts multiple joint axes, the frequencies are kept the same for all involved axes <xref ref-type="bibr" rid="B9">Davidson and Charles (2017)</xref>.</p>
</sec>
</sec>
<sec id="s2-2">
<title>2.1.3 Defining human&#x2013;exoskeleton interactions</title>
<p>Our simulation incorporates reference, tremorous, and exoskeleton-induced movements using a position-controlled upper torso model with one tremor-affected arm.</p>
<p>The exoskeleton uses an active control strategy that applies force directly to the arm. The following equations describe the process in which the force is converted into torque values that are used during the training process.</p>
<p>First, we can denote an actuator&#x2019;s state by knowing the positions of their two ends. We denote these by naming the starting point of the actuator with the number 1 and the endpoint with 2, where the actuator will exert its force and pull towards the start point.</p>
<p>The actuator force is a 3D force whose components are proportional to the angles of displacement that the two points create. With the denoted displacement angles, force components that the actuator creates on the arm at that given position can be calculated using <xref ref-type="disp-formula" rid="e1">Equations 1</xref>&#x2013;<xref ref-type="disp-formula" rid="e3">3</xref>.<disp-formula id="e1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mn>2</mml:mn>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>
<disp-formula id="e2">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mn>2</mml:mn>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>
<disp-formula id="e3">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>cos</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mn>2</mml:mn>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>z</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>To calculate the torque vectors these forces create, we first calculate the position vectors. These can be calculated with a simple vector subtraction of the point denoting the position of the specific joint (shoulder or elbow) and the endpoint of the actuator.</p>
<p>Finally, the torque values are calculated by the vector product of the force components and the position vectors and summed up for each specific joint axis. The values are then used inside the reinforcement learning environment.</p>
</sec>
<sec id="s2-3">
<title>2.1.4 The simulation system</title>
<p>For the control&#x2019;s learning loop (<xref ref-type="fig" rid="F3">Figure 3A</xref>), reference movements and the joint axes are chosen in which tremors are present. At the beginning of the episode, all actuator forces are set to 0. After summing up the actuator and tremor-generated torque values, <xref ref-type="disp-formula" rid="e4">Equation 4</xref>, a second-order, seven-variable differential equation, is solved (<xref ref-type="bibr" rid="B9">Davidson and Charles, 2017</xref>; <xref ref-type="bibr" rid="B7">Corie and Charles, 2019</xref>):<disp-formula id="e4">
<mml:math id="m5">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
<mml:mo>&#x22c5;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
<mml:mo>&#x308;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:munder accentunder="false">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
<mml:mo>&#x22c5;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
<mml:mo>&#x307;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:munder accentunder="false">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi mathvariant="bold">q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold-italic">&#x3c4;</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m6">
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the angle of displacement in each joint degree of freedom (DoF). The elements of <inline-formula id="inf3">
<mml:math id="m7">
<mml:mrow>
<mml:mi mathvariant="bold">q</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> represent the following angles: <inline-formula id="inf4">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: shoulder flexion/extension (SFE), <inline-formula id="inf5">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: shoulder abduction/adduction (SAA), <inline-formula id="inf6">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: shoulder external/internal rotation (SEIR), <inline-formula id="inf7">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: elbow flexion/extension (EFE), <inline-formula id="inf8">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>5</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: forearm pronation/supination (FPS), <inline-formula id="inf9">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: wrist flexion/extension (WFE), and <inline-formula id="inf10">
<mml:math id="m14">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: wrist radial-ulnar deviation (WRUD).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>The complete learning process of the control policy. As a deep reinforcement learning agent, we construct our controller as a multilayer perceptron (MLP) neural network. The control policy and encoder networks are updated as described in the TD7 algorithm.</p>
</caption>
<graphic xlink:href="frobt-12-1537470-g003.tif"/>
</fig>
<p>The 7 &#xd7; 7 matrices present the coupled inertia <inline-formula id="inf11">
<mml:math id="m15">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
</mml:math>
</inline-formula>, damping <inline-formula id="inf12">
<mml:math id="m16">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mi>D</mml:mi>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
</mml:math>
</inline-formula> and stiffness <inline-formula id="inf13">
<mml:math id="m17">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:munder accentunder="false">
<mml:mrow>
<mml:mi>K</mml:mi>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
<mml:mo accent="true">&#x332;</mml:mo>
</mml:munder>
</mml:mrow>
</mml:math>
</inline-formula> of the mentioned DoF, respectively <xref ref-type="bibr" rid="B9">Davidson and Charles (2017)</xref>.</p>
<p>Therefore, this takes tremor propagation into account with the inclusion of anatomically coupled properties of the joints. Furthermore, upon closer inspection, we can break down the torque values <inline-formula id="inf14">
<mml:math id="m18">
<mml:mrow>
<mml:mi mathvariant="bold-italic">&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to the following components (<xref ref-type="bibr" rid="B7">Corie and Charles, 2019</xref>): <inline-formula id="inf15">
<mml:math id="m19">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: torque required to perform an intentional task, <inline-formula id="inf16">
<mml:math id="m20">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: torque generating the tremor, <inline-formula id="inf17">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: task load torque, and <inline-formula id="inf18">
<mml:math id="m22">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: gravitational torque, <inline-formula id="inf19">
<mml:math id="m23">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>O</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: torque generated by the orthosis (exoskeleton) on the particular joint.</p>
<p>The mentioned components <inline-formula id="inf20">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf21">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf22">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>G</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are covered by the reference movement, thus leaving the torque generated by the tremor and the exoskeleton to find the unknown joint angle displacement values in our calculation.</p>
<p>With the calculated joint angle values based on the reference motion recording and tremor&#x2013;exoskeleton interaction, we simulate one step in our simulation and receive a new state observation.</p>
<p>This new state observation is then propagated through an encoder neural network to further extract hidden information or unrealized correlations in the data, which are then given as the input to the control policy alongside the original observation vector that the encoder received.</p>
<p>For the anatomical properties of the simulation, the joint angle ranges are based on <xref ref-type="bibr" rid="B67">Zwerus et al. (2019)</xref> and <xref ref-type="bibr" rid="B21">Gill et al. (2020)</xref>. The maximum joint torques for the voluntary motion are designed according to <xref ref-type="bibr" rid="B43">Otis et al. (1990)</xref> and <xref ref-type="bibr" rid="B22">G&#xfc;nzkofer et al. (2012)</xref>. The upper and lower arm weight ratios are defined by <xref ref-type="bibr" rid="B47">Plagenhoef et al. (1983)</xref>.</p>
<sec id="s2-3-1">
<title>2.1.5 Dynamics randomization</title>
<p>Although simulation-based training provides a safe and efficient way to train our controller, there is a well-known discrepancy called the sim-to-real gap between the physical and real-world environments.</p>
<p>In order to overcome this obstacle and improve the robustness of our control, we employ dynamics randomization (<xref ref-type="bibr" rid="B51">Sadeghi and Levine, 2016</xref>; <xref ref-type="bibr" rid="B59">Tobin et al., 2017</xref>; <xref ref-type="bibr" rid="B46">Peng et al., 2018b)</xref>.</p>
<p>This method randomly samples environmental characteristics from a given uniform distribution (<xref ref-type="table" rid="T1">Table 1</xref>) at the beginning of each episode. This forces our agent to be more robust against perturbations present in the environmental characteristics and to better adapt to the real-world environment, whose characteristics are expected to be present in the given distribution ranges.</p>
</sec>
</sec>
<sec id="s2-4">
<title>2.2 Control algorithm training</title>
<p>In this section, we propose a deep reinforcement learning-based training and testing framework that enables the learning of optimal tremor-suppression strategy.</p>
<sec id="s2-4-1">
<title>2.2.1 Reinforcement learning background</title>
<p>Reinforcement learning (RL) is a branch of machine learning that deals with sequential decision-making problems (<xref ref-type="bibr" rid="B57">Sutton and Barto, 2018</xref>). The objective is to learn an optimal policy <inline-formula id="inf23">
<mml:math id="m27">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> that enables an agent to maximize its return through interactions with a specified environment. The return, which is described as the discounted cumulative rewards the agent collects, is defined by <xref ref-type="disp-formula" rid="e5">Equation 5</xref>.<disp-formula id="e5">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x22c5;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>
</p>
<p>The agent at each discrete time step <inline-formula id="inf24">
<mml:math id="m29">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, with a corresponding state <inline-formula id="inf25">
<mml:math id="m30">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, selects an action <inline-formula id="inf26">
<mml:math id="m31">
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> with respect to its policy <inline-formula id="inf27">
<mml:math id="m32">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
<mml:mo>:</mml:mo>
<mml:mi>S</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, receiving reward <inline-formula id="inf28">
<mml:math id="m33">
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> and a new state of the environment <inline-formula id="inf29">
<mml:math id="m34">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Deep reinforcement learning is the combination of deep neural networks with RL, where the policy is represented by a neural network <inline-formula id="inf30">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula id="inf31">
<mml:math id="m36">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> denotes the weights of the network.</p>
</sec>
<sec id="s2-4-2">
<title>2.2.2 TD7</title>
<p>A popular family of RL methods is actor-critic algorithms, where a policy known as the actor is updated by the deterministic policy gradient algorithm (<xref ref-type="bibr" rid="B53">Silver et al., 2014</xref>) <xref ref-type="disp-formula" rid="e6">Equation 6</xref>:<disp-formula id="e6">
<mml:math id="m37">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>J</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>&#x2207;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(6)</label>
</disp-formula>
</p>
<p>In <xref ref-type="disp-formula" rid="e7">Equation 7</xref>, <inline-formula id="inf32">
<mml:math id="m38">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is known as the critic or value function, which is used to calculate the expected return when performing action <inline-formula id="inf33">
<mml:math id="m39">
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> in a given state <inline-formula id="inf34">
<mml:math id="m40">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> following the actor policy <inline-formula id="inf35">
<mml:math id="m41">
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.<disp-formula id="e7">
<mml:math id="m42">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x223c;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(7)</label>
</disp-formula>which is commonly updated by temporal difference learning utilizing a secondary target network as described by <xref ref-type="bibr" rid="B41">Mnih et al. (2013)</xref> see <xref ref-type="disp-formula" rid="e8">Equation 8</xref>.<disp-formula id="e8">
<mml:math id="m43">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="2em"/>
<mml:msup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(8)</label>
</disp-formula>where <inline-formula id="inf36">
<mml:math id="m44">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the target critic network, and <inline-formula id="inf37">
<mml:math id="m45">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the target actor network.</p>
<p>These methods are prone to overestimation errors, whereby, through the function approximation error of the critic, some state-value pairs are overestimated, leading to a sub-optimal policy. The twin delayed deep deterministic policy algorithm (TD3) (<xref ref-type="bibr" rid="B19">Fujimoto et al., 2018</xref>) solves function approximation errors by the use of a second critic network and clipped doubled Q-learning (<xref ref-type="bibr" rid="B60">Van Hasselt et al., 2016</xref>), as shown in <xref ref-type="disp-formula" rid="e9">Equation 9</xref>.<disp-formula id="e9">
<mml:math id="m46">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1,2</mml:mn>
</mml:mrow>
</mml:munder>
<mml:msub>
<mml:mrow>
<mml:mi>Q</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c0;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(9)</label>
</disp-formula>
</p>
<p>In our proposed reinforcement learning-based controller, a state-of-the-art reinforcement learning algorithm called TD7 (<xref ref-type="bibr" rid="B17">Fujimoto et al., 2023</xref>), which incorporates the following additions to the TD3 algorithm, is used.</p>
<p>A loss-adjusted prioritized (LAP) (<xref ref-type="bibr" rid="B20">Fujimoto et al., 2020</xref>) replay buffer improves the sample efficiency of the algorithm and speeds up training by sampling transition tuples <inline-formula id="inf38">
<mml:math id="m47">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2254;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> from which the agent can learn more. The probability of sampling transition <inline-formula id="inf39">
<mml:math id="m48">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> from the replay buffer <inline-formula id="inf40">
<mml:math id="m49">
<mml:mrow>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> sampling is<disp-formula id="e10">
<mml:math id="m50">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>max</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mspace width="2em"/>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mspace width="0.3333em"/>
<mml:mi>&#x3b4;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b3;</mml:mi>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(10)</label>
</disp-formula>
</p>
<p>In <xref ref-type="disp-formula" rid="e10">Equation 10</xref>, the level of prioritization is governed by the hyperparameter <inline-formula id="inf41">
<mml:math id="m51">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>Behavioral cloning term allows the use of the algorithm in an offline-RL setting <xref ref-type="bibr" rid="B18">Fujimoto and Gu (2021)</xref>. However, because our task relies on online training, we do not go into depth for this addition.</p>
<p>Policy checkpoints add additional stability toward the training of the agent by selectively employing the best-performing networks, therefore providing stability.</p>
<p>State-action learned embeddings aim to improve the inputs to the actor and critic networks by capturing the relevant underlying structure of the observation space and the transition dynamics present in the environment. Therefore, our network equations can be described by <xref ref-type="disp-formula" rid="e11">Equation 11</xref> as follows:<disp-formula id="e11">
<mml:math id="m52">
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="2em"/>
<mml:mi>&#x3c0;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>&#x3c0;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(11)</label>
</disp-formula>
</p>
<p>where <inline-formula id="inf42">
<mml:math id="m53">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the state embedding, and <inline-formula id="inf43">
<mml:math id="m54">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>z</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> refers to the state-action embedding.</p>
<p>The choice of TD7 (<xref ref-type="bibr" rid="B17">Fujimoto et al., 2023</xref>) over other widely used reinforcement learning algorithms such as PPO (<xref ref-type="bibr" rid="B52">Schulman et al., 2017</xref>), TD3 (<xref ref-type="bibr" rid="B19">Fujimoto et al., 2018</xref>), or SAC (<xref ref-type="bibr" rid="B24">Haarnoja et al., 2018</xref>) is motivated by several key factors. First, TD7 exhibits significantly improved sample efficiency, often achieving comparable performance to prior methods with only one-tenth of the training time steps. Second, it demonstrates substantially higher performance across standard gym benchmark tasks (<xref ref-type="bibr" rid="B5">Brockman et al., 2016</xref>). Finally, TD7 incorporates embeddings that enable the use of larger neural network architectures. A detailed list of hyperparameters with their justifications is provided in the supplementary material.</p>
</sec>
<sec id="s2-4-3">
<title>2.2.3 Observations, actions, and rewards</title>
<p>At each time step <inline-formula id="inf44">
<mml:math id="m55">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, an observation vector of <inline-formula id="inf45">
<mml:math id="m56">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>o</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf46">
<mml:math id="m57">
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>80</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. The observation/state vector is defined by <inline-formula id="inf47">
<mml:math id="m58">
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>:</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, in which <inline-formula id="inf48">
<mml:math id="m59">
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> contains the force values of the actuator, <inline-formula id="inf49">
<mml:math id="m60">
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> refers to the tremor torque, <inline-formula id="inf50">
<mml:math id="m61">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> contain the end position coordinates of the actuators, and <inline-formula id="inf51">
<mml:math id="m62">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mi>p</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the coordinate positions of the joints. In this observation vector, all the values are normalized.</p>
<p>For each observation vector, the actor network outputs an action <inline-formula id="inf52">
<mml:math id="m63">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> <inline-formula id="inf53">
<mml:math id="m64">
<mml:mrow>
<mml:mo>&#x2208;</mml:mo>
<mml:mspace width="0.3333em"/>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="double-struck">R</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>7</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> in the form of the output force of each actuator. These actions are then converted into the ranges of the actuator forces <italic>F</italic>.</p>
<p>To achieve the complex tremor-suppression behavior of our agent, a densely constructed reward function is utilized. The aim of the control is to suppress tremors to the maximum extent while interfering the least with the voluntary movement of the patient and also utilizing the minimum force required.</p>
<p>Therefore, the reward function consists of five parts: a part accounting for mitigating the tremor torque, a sub-reward accounting for the distortion of the original movement trajectory, a term encouraging tremor reduction across all the affected axes, an actuator smoothness reward, and a reward encouraging the use of minimal actuator force in order to control this tremor torque. This reward is based on the reinforcement learning heuristics of reward shaping (<xref ref-type="bibr" rid="B45">Peng et al., 2018a)</xref>. The full reward function is written as <xref ref-type="disp-formula" rid="e12">Equation 12</xref>:<disp-formula id="e12">
<mml:math id="m65">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
<label>(12)</label>
</disp-formula>where <inline-formula id="inf54">
<mml:math id="m66">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf55">
<mml:math id="m67">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf56">
<mml:math id="m68">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf57">
<mml:math id="m69">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are the respective weights of the sub-rewards. The values of the weights are <inline-formula id="inf58">
<mml:math id="m70">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf59">
<mml:math id="m71">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.9</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf60">
<mml:math id="m72">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.05</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula id="inf61">
<mml:math id="m73">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.05</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, and <inline-formula id="inf62">
<mml:math id="m74">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. <inline-formula id="inf63">
<mml:math id="m75">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf64">
<mml:math id="m76">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the maximum actuator forces possible at the shoulder and elbow actuators.</p>
<p>The tremor axis reward <inline-formula id="inf65">
<mml:math id="m77">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> aims to encourage control strategies that suppress tremors across all the involved joint axes, as defined in <xref ref-type="disp-formula" rid="e13">Equation 13</xref>:<disp-formula id="e13">
<mml:math id="m78">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
<label>(13)</label>
</disp-formula>where <inline-formula id="inf66">
<mml:math id="m79">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of axes where generated tremor torques are present.</p>
<p>The torque reward <inline-formula id="inf67">
<mml:math id="m80">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula> enforces the agent to mitigate tremors in all the affected joint axes:<disp-formula id="e14">
<mml:math id="m81">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
</mml:mrow>
</mml:mfenced>
<mml:mo>/</mml:mo>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(14)</label>
</disp-formula>
</p>
<p>
<xref ref-type="disp-formula" rid="e14">Equation 14</xref> contains the unmitigated original tremor-generated torque values <inline-formula id="inf68">
<mml:math id="m82">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and the torque values after the exoskeleton has applied its forces <inline-formula id="inf69">
<mml:math id="m83">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. The equation calculates the tremor suppression on a given joint axis, which is then averaged to be capable of handling tremors affecting multiple joint axes.</p>
<p>The actuator force reward encourages the agent to apply minimal forces with the exoskeleton actuators, reducing energy expenditure, improving efficiency, and preventing damage to the exoskeleton and the patient.<disp-formula id="e15">
<mml:math id="m84">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(15)</label>
</disp-formula>
</p>
<p>In <xref ref-type="disp-formula" rid="e15">Equation 15</xref>, <inline-formula id="inf70">
<mml:math id="m85">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>w</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is an actuator weight aimed to magnify the learning signal, whose value is dependent on the highest maximum force output and the number of actuators present in the exoskeleton.</p>
<p>The action smoothness reward <xref ref-type="disp-formula" rid="e16">Equation 16</xref> promotes the use of smooth actuator forces by penalizing the second-order derivatives of the actuator forces:<disp-formula id="e16">
<mml:math id="m86">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x22c5;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:munderover>
</mml:mstyle>
<mml:msup>
<mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#x22c5;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
<label>(16)</label>
</disp-formula>
</p>
<p>Because exoskeletons can disrupt natural movements, the &#x201c;unwanted movement&#x201d; reward <xref ref-type="disp-formula" rid="e17">Equation 17</xref> is added to ensure smoother, more natural motion, minimizing discomfort and improving efficiency. The reward discourages the control from interfering with the voluntary movement trajectory by penalizing the amount of torque created on non-tremor-affected axes.<disp-formula id="e17">
<mml:math id="m87">
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mo>&#x2211;</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>u</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>F</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
<label>(17)</label>
</disp-formula>
</p>
</sec>
<sec id="s2-4-4">
<title>2.2.4 Modifications to handle tremor suppression</title>
<p>Given the diverse nature and precision demands inherent in the tremor-suppression task, the training algorithm has undergone specific modifications to accommodate these challenges.</p>
<p>First, a modification is made to the replay buffer to handle the variance in movement trajectories present in the reference movement. This way, the replay buffer is divided into as many sub-parts as there are reference movements, and then from these sub-buffers, we sample a batch size number of transitions according to prioritized experience replay (<xref ref-type="bibr" rid="B20">Fujimoto et al., 2020</xref>). This modification improves robustness because oversampling is avoided even though the different length reference movements create an uneven data distribution in the buffer overall. The sub-buffer also has an increased size to leverage a wider range of possible transitions present to improve the performance of training (<xref ref-type="bibr" rid="B16">Fedus et al., 2020</xref>).</p>
<p>The other main modification is regarding the decrease of action and policy noise in the algorithm. This helps by reducing the random space around the agent&#x2019;s chosen action/policy values, therefore allowing it to learn more fine-tuned control policies. This is crucial because a small change in actuator force can lead to vast differences in the torque created on the human skeleton due to anatomical reasons.</p>
<p>Tremor suppression via exoskeleton requires sophisticated actuation of different motors, where we found that typical white noise exploration added to the chosen actions is not sufficient. Therefore, we replace this common method by adding pink noise (<xref ref-type="bibr" rid="B14">Eberhard et al., 2023</xref>) to the actions, improving the agent&#x2019;s exploration ability and improving action smoothness by incorporating a more correlated noise to the actions.</p>
</sec>
<sec id="s2-4-5">
<title>2.2.5 Training details</title>
<p>The training of the agent is performed in one set of parallel environments, where each represents a reference movement trajectory and a distinctly generated tremor, with the axes defined where tremors are present. The axes in which the tremors are present are constant throughout all the dynamic movements. The agent does not use random state initialization or early termination, but it ensures that the simulated trajectory remains close to the original trajectory by initializing each simulation step from the original value of the reference movement and not the previous simulation step positions. This ensures robustness and boosts performance.</p>
<p>The specifics of the networks and hyperparameter details of the training are found in the supplementary materials.</p>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<p>To analyze our control algorithm&#x2019;s performance, a number of numerical tests were conducted to answer the following questions: 1) Can the trained agent suppress the generated tremors across various joints and reference movement? 2) How accurately can it mitigate the effects of tremors, and at what percentage? 3) To what extent are the generated tremor torque values suppressed? 4) How is the original movement trajectory affected by the exoskeleton?</p>
<sec id="s3-1">
<title>3.1 Evaluation of the control policy</title>
<p>The control policy was evaluated through 100 episodes, each of which consisted of an environment with each of the reference movements. The environment characteristics were sampled from a larger dynamics randomization testing range (<xref ref-type="table" rid="T1">Table 1</xref>) to display the learned controller&#x2019;s ability to generalize to out-of-distribution cases. The frequency components for the tremors were randomly generated in the specified range in each episode and environment. The control has been trained and evaluated for each possible tremor combination involving the shoulder axes and the elbow extension/flexion axis. The tremor amplitude suppression percentages and the occurrences of tremor suppression without interfering with the original movement trajectory can be seen in <xref ref-type="fig" rid="F4">Figure 4</xref>. The controller effectively suppresses tremors in all but one of the generated tremor pairs, demonstrating a high level of generalizability of the method. In-depth performance data for each of the combinations of the tremor joint axes are presented in the supplementary material.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Dynamic randomization parameter ranges used throughout training and validation. Anatomical matrices represent values in the inertia, damping, and stiffness matrices. Actuator precision accounts for the discrepancy between the commanded and actual force generated by the actuator. Actuator end-point shift refers to actuator sliding due to soft-robotic Velcro changes. Tremor frequencies and amplitude reflect different Parkinson&#x2019;s patients&#x2019; tremor characteristics.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left">Dynamics parameters</th>
<th align="left">Training range</th>
<th align="left">Testing range</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Anatomical matrices</td>
<td align="left">[0.9, 1.1] <inline-formula id="inf71">
<mml:math id="m88">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> original value</td>
<td align="left">[0.875, 1.125] <inline-formula id="inf72">
<mml:math id="m89">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> original value</td>
</tr>
<tr>
<td align="left">Actuator precision</td>
<td align="left">[0.97, 1.03] <inline-formula id="inf73">
<mml:math id="m90">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> original value</td>
<td align="left">[0.96, 1.04] <inline-formula id="inf74">
<mml:math id="m91">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> original value</td>
</tr>
<tr>
<td align="left">Actuator end-point shift (in one axis)</td>
<td align="left">[0, 2] cm</td>
<td align="left">[0, 2.5] cm</td>
</tr>
<tr>
<td align="left">Tremor first harmonic frequency</td>
<td align="left">[4, 6] Hz</td>
<td align="left">[3.75, 6.25] Hz</td>
</tr>
<tr>
<td align="left">Tremor second harmonic frequency</td>
<td align="left">[8, 12] Hz</td>
<td align="left">[7.5, 12.5] Hz</td>
</tr>
<tr>
<td align="left">Tremor amplitude</td>
<td align="left">[0.1, 1] <inline-formula id="inf75">
<mml:math id="m92">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> max value</td>
<td align="left">[0.05, 1.05] <inline-formula id="inf76">
<mml:math id="m93">
<mml:mrow>
<mml:mo>&#x22c5;</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> max value</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Tremor amplitude suppression values throughout the tremor pairs. Tremor suppression was evaluated for each case over 100 episodes across all reference movements, averaging tremor amplitude suppression and occurrence values. Tremor occurrence indicates the percentage of time steps where the tremor was reduced without disrupting the person&#x2019;s original trajectory.</p>
</caption>
<graphic xlink:href="frobt-12-1537470-g004.tif"/>
</fig>
<p>To further investigate the reference motion-wise performance of the controller, we evaluate a tremor case involving the elbow flexion/extension axis of the arm. The performance of the control can be seen in <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Tremor amplitude suppression values throughout the movements. We display the tremor amplitude suppression values achieved by the exoskeleton throughout the recorded dynamic movement over the time steps of a single episode. The reported statistics are computed over 100 episodes with out-of-training distribution domain randomization.</p>
</caption>
<graphic xlink:href="frobt-12-1537470-g005.tif"/>
</fig>
<p>These results demonstrate that a single RL-based trained controller can adapt to mitigate tremors regardless of the reference movement. The control also displays high performance, with the maximum tremor amplitude suppression values exceeding 99%. The control also displays a consistent ability to suppress tremors, evident from the median and mean values of <xref ref-type="fig" rid="F5">Figure 5</xref>.</p>
<p>The qualitative performance of the controller can be seen in <xref ref-type="fig" rid="F6">Figure 6A</xref>. The figure shows how the original movement trajectory is affected by the exoskeleton. This figure presents additional evidence, as the suppressed movement trajectory consistently maintains a shorter distance from the reference movement&#x2019;s trajectory when compared to the trajectory affected by tremors.</p>
<p>The torque plots in <xref ref-type="fig" rid="F6">Figures 6B&#x2013;D</xref> display the controller-created torque present on the joints unaffected by the simulated tremor. The controller&#x2019;s ideal behavior, which is derived from the torque values achieving the optimal zero generated torque at given time steps, can be seen in the figures. Furthermore, when the torque values are not 0, we can see a tendency in the time steps to minimize this torque and correct the control behavior.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>The trajectories of the simulation. <bold>(A)</bold> The trajectories in the 3D simulation environment. Trajectories were recorded during an external rotation movement of the shoulder. Tremors were observed in the flexion/extension axis of the elbow joint. The blue represents the original trajectory, the green represents the trajectory affected by the tremor, and the red represents the exoskeleton-suppressed trajectory. <bold>(B &#x2d;E)</bold> The suppressed and unsuppressed torque values present at each joint.</p>
</caption>
<graphic xlink:href="frobt-12-1537470-g006.tif"/>
</fig>
<p>The torque plot in <xref ref-type="fig" rid="F6">Figure 6E</xref> shows the torque created by the controllers present in the joint affected by the simulated tremor. The trained controller effectively suppresses tremors in the involved axis in which tremors are present, although it is prone to producing different torque suppression percentages. The concrete torque suppression values for each tremor pair averaged across the movements are provided in the supplementary material.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Discussion and limitations</title>
<p>The control of soft-robotic exoskeletons requires real-time decision making on a wide range of stochastic predictors and changing sensory readings in dynamic everyday movements. Furthermore, validation of these control algorithms requires extensive testing to ensure the safety and performance of the control.</p>
<p>This work contains an easily adaptable testing environment for all sorts of neurodegenerative diseases displaying symptoms of tremors, allowing for rapid and cheap testing for new learning-based control methods.</p>
<p>The controller also displays good performance in mitigating tremor torques and amplitudes. However, this performance fluctuates over time steps. The controller&#x2019;s performance varies between movements and tremor cases. This is due to the exoskeleton structure and also the simulation tremor torque values used. In future studies, exploration of tremor torque ranges is warranted. Furthermore, addressing performance optimization requires a more nuanced understanding of the involved joint axes and their associated characteristics in a given dynamic movement.</p>
<p>From the tremor torque plots, it is evident that further reduction in the tremor amplitude is dependent on how effectively the exoskeleton only exerts forces onto the axes where tremors are present. This could involve revisiting the actuator positions or perhaps using a mixed method of FES and robotic exoskeletons. This approach could minimize tremors in the shoulder flexion/extension axis, a case that, along with its variants, achieved the lowest tremor suppression torque/amplitude values.</p>
<p>This control uses state-of-the-art approaches introduced by encoder networks to extract information from changing observations induced by tremors to achieve high-performance tremor suppression. The achieved performances also highlight the need for improved neural network architectures in these algorithms to improve the safety and stability of these control methods, which cannot be bypassed by hybrid traditional learning-based control approaches because the neural network actor&#x2019;s densely connected architecture can generate values vastly different in concurrent time steps. These generated values cannot be mitigated meaningfully by a traditional proportional-integrative-derivative (PID) controller. This is a future challenge to be addressed due to the limitations of the frequency of control actuation usually present (30&#x2013;40 Hz) in the actuators.</p>
<p>The results show promise, but the current research is limited to simulation. We mitigate this limitation through domain randomization methods as much as possible. Furthermore, as our training algorithm relies on Markov decision process (MDP) formulation, additional considerations must be made to maintain accurate sensor readings by either state estimation techniques such as Kalman filters (<xref ref-type="bibr" rid="B30">Kalman, 1960</xref>) or by incorporating an algorithm that can handle partially observable states. Movements that differ significantly from the reference trajectories used in training may limit the controller&#x2019;s accuracy. Consequently, future work should include experimental validation on patients and testing with out-of-distribution simulation movements to fully understand how this impacts the controller&#x2019;s performance. For safety reasons, built-in safety checks, torque limits, and acceleration thresholds should be incorporated into the deployed exoskeleton to further mitigate this problem.</p>
<p>Current applications of the exoskeleton control system extend to real-life rehabilitation exercises, similar to the trajectories present during training.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>This paper proposes a physical simulation-based tremor-suppressing exoskeleton physical simulation framework. The framework is flexible and adaptable to different diseases and characteristics of patients with tremor symptoms. It can also be incorporated with various other dynamic movements. This simulation also proposes an inexpensive and rapid method of validating control algorithm performances. The simulation hyperparameters are included in the supplementary material.</p>
<p>The paper also details the training of a reinforcement learning-based encoder-actor controller. The controller can adapt to personalized interventions in the management of movement disorders. Additionally, the controller can adjust to varying ranges of actuator forces, thereby proposing a viable strategy for tremor suppression.</p>
<p>Experimental results show that the proposed framework can mitigate tremor torques present at the joint axes, and the entire tremor amplitude with tremor propagation is taken into account. The results indicate a substantial decrease in both median and maximum tremor amplitudes.</p>
<p>The control aims to mitigate tremors without interfering with the original movement, not allowing patients to rely too heavily on the exoskeleton during natural motor abilities, thereby not hindering rehabilitation efforts while also minimizing the potential side effects that might arise from prolonged use.</p>
<p>In the future, we intend to deploy the trained exoskeleton control on physical hardware, incorporating sim-to-real techniques into the physical simulation. Furthermore, we aim to validate the performance of this control in a clinical trial setting with patients involved.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The raw data reference motion recordings in the article will be made available by the authors, without undue reservation. Also the log files of the evaluation process will be made available to ensure the accuracy and transparency of the research. The modified version of the TD7 algorithm is open sourced: (<ext-link ext-link-type="uri" xlink:href="https://github.com/TomasDelaney/A-Deep-Reinforcement-Learning-Enabled-Soft-Exoskeleton-for-Parkinson-s-Patients">https://github.com/TomasDelaney/A-Deep-Reinforcement-Learning-Enabled-Soft-Exoskeleton-for-Parkinson-s-Patients</ext-link>).</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The study protocol was approved by ETT TUKEB (reference number: IV/8514-3/2021/EKU), and the study was conducted in accordance with the Declaration of Helsinki. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>TE: data curation, investigation, methodology, software, visualization, writing&#x2013;original draft, and writing&#x2013;review and editing. SF: conceptualization, project administration, and writing&#x2013;review and editing. &#xc1;M: conceptualization, project administration, and writing&#x2013;review and editing. GC: conceptualization, funding acquisition, project administration, supervision, writing&#x2013;review and editing, and writing&#x2013;original draft.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. Project no. 2020-1.1.5-GYORS&#xcd;T&#xd3;S&#xc1;V-2021-00022 has been implemented with the support provided by the Ministry of Culture and Innovation of Hungary from the National Research, Development and Innovation Fund, financed under the 2020-1.1.5-GYORS&#xcd;T&#xd3;S&#xc1;V funding scheme. Additional support came from the &#xda;NKP-23-1-I-PPKE-5 national excellence program and the TKP-2021_02-NVA-27 grant from the National Research, Development and Innovation Office.</p>
</sec>
<ack>
<p>The authors would like to acknowledge the insight, opportunity, and help provided by the Pet&#x151; Institute. The authors are grateful to Edward and Martha Kovach for their manuscript writing suggestions.</p>
</ack>
<sec sec-type="COI-statement" id="s10">
<title>Conflict of interest</title>
<p>Authors TE, SF and GC were employed by Jedlik Innovation Ltd.</p>
<p>The remaining author declares that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s11">
<title>Generative AI statement</title>
<p>The author(s) declare that Generative AI was used in the creation of this manuscript. To find catchy title suggestions and for sentence-level grammar editing.</p>
</sec>
<sec sec-type="disclaimer" id="s12">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors, and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s13">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frobt.2025.1537470/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frobt.2025.1537470/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>Reference</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Albano</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Basaia</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Emedoli</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Balestrino</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Pompeo</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Barzaghi</surname>
<given-names>L. R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Longitudinal brain functional connectivity changes induced by neurosurgical thalamotomy for tremor in Parkinson&#x2019;s disease: a preliminary study</article-title>. <source>J. Neurology</source> <volume>270</volume>, <fpage>3623</fpage>&#x2013;<lpage>3629</lpage>. <pub-id pub-id-type="doi">10.1007/s00415-023-11705-2</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Asil</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Ahlawat</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Barroso</surname>
<given-names>G. G.</given-names>
</name>
<name>
<surname>Narayan</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Nanomaterial based drug delivery systems for the treatment of neurodegenerative diseases</article-title>. <source>Biomaterials Sci.</source> <volume>8</volume>, <fpage>4109</fpage>&#x2013;<lpage>4128</lpage>. <pub-id pub-id-type="doi">10.1039/d0bm00809e</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Awantha</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Wanasinghe</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kavindya</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kulasekera</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Chathuranga</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>A novel soft glove for hand tremor suppression: evaluation of layer jamming actuator placement</article-title>,&#x201d; in <source>2020 3rd IEEE international conference on soft robotics (RoboSoft)</source> (<publisher-name>IEEE</publisher-name>), <fpage>440</fpage>&#x2013;<lpage>445</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhatia</surname>
<given-names>K. P.</given-names>
</name>
<name>
<surname>Bain</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Bajaj</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Elble</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Hallett</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Louis</surname>
<given-names>E. D.</given-names>
</name>
<etal/>
</person-group> (<year>2018</year>). <article-title>Consensus statement on the classification of tremors. from the task force on tremor of the international Parkinson and movement disorder society</article-title>. <source>Mov. Disord.</source> <volume>33</volume>, <fpage>75</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1002/mds.27121</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brockman</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Cheung</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Pettersson</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tang</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2016</year>). <article-title>Openai gym</article-title>. <comment>
<italic>arXiv preprint arXiv:1606.01540</italic>
</comment>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zeng</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Sun</surname>
<given-names>T.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Role of traditional Chinese medicine in ameliorating mitochondrial dysfunction via non-coding rna signaling: implication in the treatment of neurodegenerative diseases</article-title>. <source>Front. Pharmacol.</source> <volume>14</volume>, <fpage>1123188</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2023.1123188</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Corie</surname>
<given-names>T. H.</given-names>
</name>
<name>
<surname>Charles</surname>
<given-names>S. K.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Simulated tremor propagation in the upper limb: from muscle activity to joint displacement</article-title>. <source>J. biomechanical Eng.</source> <volume>141</volume>, <fpage>0810011</fpage>&#x2013;<lpage>08100117</lpage>. <pub-id pub-id-type="doi">10.1115/1.4043442</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Coumans</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Pybullet, a python module for physics simulation for games, robotics and machine learning</article-title>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="http://pybullet.org">http://pybullet.org</ext-link>
</comment>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davidson</surname>
<given-names>A. D.</given-names>
</name>
<name>
<surname>Charles</surname>
<given-names>S. K.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Fundamental principles of tremor propagation in the upper limb</article-title>. <source>Ann. Biomed. Eng.</source> <volume>45</volume>, <fpage>1133</fpage>&#x2013;<lpage>1147</lpage>. <pub-id pub-id-type="doi">10.1007/s10439-016-1765-5</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Deuschl</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bain</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Brin</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Committee</surname>
<given-names>A. H. S.</given-names>
</name>
</person-group> (<year>1998</year>). <article-title>Consensus statement of the movement disorder society on tremor</article-title>. <source>Mov. Disord.</source> <volume>13</volume>, <fpage>2</fpage>&#x2013;<lpage>23</lpage>. <pub-id pub-id-type="doi">10.1002/mds.870131303</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dideriksen</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Laine</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Dosen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Muceli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rocon</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pons</surname>
<given-names>J. L.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Electrical stimulation of afferent pathways for the suppression of pathological tremor</article-title>. <source>Front. Neurosci.</source> <volume>11</volume>, <fpage>178</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2017.00178</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ding</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Kuindersma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Walsh</surname>
<given-names>C. J.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Human-in-the-loop optimization of hip assistance with a soft exosuit during walking</article-title>. <source>Sci. robotics</source> <volume>3</volume>, <fpage>eaar5438</fpage>. <pub-id pub-id-type="doi">10.1126/scirobotics.aar5438</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dosen</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Muceli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dideriksen</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Romero</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Rocon</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Pons</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>Online tremor suppression using electromyography and low-level electrical stimulation</article-title>. <source>IEEE Trans. Neural Syst. Rehabilitation Eng.</source> <volume>23</volume>, <fpage>385</fpage>&#x2013;<lpage>395</lpage>. <pub-id pub-id-type="doi">10.1109/tnsre.2014.2328296</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Eberhard</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Hollenstein</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Pinneri</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Martius</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Pink noise is all you need: colored noise exploration in deep reinforcement learning</article-title>,&#x201d; in <conf-name>The Eleventh International Conference on Learning Representations</conf-name>, <conf-loc>China</conf-loc>, <conf-date>May 1 &#x2014; Fri May 5</conf-date>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Faraji</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Rouhollahi</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Paghaleh</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Gheisarnejad</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Khooban</surname>
<given-names>M.-H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Adaptive multi symptoms control of Parkinson&#x2019;s disease by deep reinforcement learning</article-title>. <source>Biomed. Signal Process. Control</source> <volume>80</volume>, <fpage>104410</fpage>. <pub-id pub-id-type="doi">10.1016/j.bspc.2022.104410</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fedus</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Ramachandran</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Agarwal</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Larochelle</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Rowland</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Revisiting fundamentals of experience replay</article-title>
</citation>
</ref>
<ref id="B17">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Fujimoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>W.-D.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>S. S.</given-names>
</name>
<name>
<surname>Precup</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Meger</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>For sale: state-action representation learning for deep reinforcement learning</article-title>. <source>Advances in neural information processing systems</source>. <volume>36</volume>, <fpage>61573</fpage>&#x2013;<lpage>61624</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fujimoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Gu</surname>
<given-names>S. S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A minimalist approach to offline reinforcement learning</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>34</volume>, <fpage>20132</fpage>&#x2013;<lpage>20145</lpage>.</citation>
</ref>
<ref id="B19">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fujimoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hoof</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Meger</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Addressing function approximation error in actor-critic methods</article-title>,&#x201d; in <conf-name>International conference on machine learning</conf-name> <conf-loc>USA</conf-loc>, <conf-date>2025 &#x2013; Sat, 19 Jul, 2025</conf-date>, (<publisher-name>PMLR</publisher-name>), <fpage>1587</fpage>&#x2013;<lpage>1596</lpage>.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fujimoto</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Meger</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Precup</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An equivalence between loss functions and non-uniform sampling in experience replay</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>14219</fpage>&#x2013;<lpage>14230</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gill</surname>
<given-names>T. K.</given-names>
</name>
<name>
<surname>Shanahan</surname>
<given-names>E. M.</given-names>
</name>
<name>
<surname>Tucker</surname>
<given-names>G. R.</given-names>
</name>
<name>
<surname>Buchbinder</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Hill</surname>
<given-names>C. L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Shoulder range of movement in the general population: age and gender stratified normative data using a community-based cohort</article-title>. <source>BMC Musculoskelet. Disord.</source> <volume>21</volume>, <fpage>676</fpage>&#x2013;<lpage>679</lpage>. <pub-id pub-id-type="doi">10.1186/s12891-020-03665-9</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>G&#xfc;nzkofer</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Bubb</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Bengler</surname>
<given-names>K.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>Maximum elbow joint torques for digital human models</article-title>. <source>Int. J. Hum. Factors Model. Simul.</source> <volume>3</volume>, <fpage>109</fpage>&#x2013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.1504/ijhfms.2012.051092</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haarnoja</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Moran</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lever</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S. H.</given-names>
</name>
<name>
<surname>Tirumala</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Humplik</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Learning agile soccer skills for a bipedal robot with deep reinforcement learning</article-title>. <source>Science Robotics</source> <volume>9</volume> (<issue>89</issue>), <fpage>eadi8022</fpage>.</citation>
</ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Haarnoja</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Soft actor-critic: off-policy maximum entropy deep reinforcement learning with a stochastic actor</article-title>,&#x201d; in <conf-name>International conference on machine learning</conf-name> <conf-loc>China</conf-loc>, <conf-date>19 Jul, 2025</conf-date>, (<publisher-name>Pmlr</publisher-name>), <fpage>1861</fpage>&#x2013;<lpage>1870</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heemels</surname>
<given-names>M.-T.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Neurodegenerative diseases</article-title>. <source>Nature</source> <volume>539</volume>, <fpage>179</fpage>&#x2013;<lpage>180</lpage>. <pub-id pub-id-type="doi">10.1038/539179a</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Heris</surname>
<given-names>R. M.</given-names>
</name>
<name>
<surname>Shirvaliloo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Abbaspour-Aghdam</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Hazrati</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Shariati</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Youshanlouei</surname>
<given-names>H. R.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>The potential use of mesenchymal stem cells and their exosomes in Parkinson&#x2019;s disease treatment</article-title>. <source>Stem Cell. Res. and Ther.</source> <volume>13</volume>, <fpage>371</fpage>. <pub-id pub-id-type="doi">10.1186/s13287-022-03050-4</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Herrnstadt</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Menon</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Voluntary-driven elbow orthosis with speed-controlled tremor suppression</article-title>. <source>Front. Bioeng. Biotechnol.</source> <volume>4</volume>, <fpage>29</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2016.00029</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jin</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Shao</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>High-speed quadrupedal locomotion by imitation-relaxation reinforcement learning</article-title>. <source>Nat. Mach. Intell.</source> <volume>4</volume>, <fpage>1198</fpage>&#x2013;<lpage>1208</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-022-00576-3</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jitkritsadakul</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Thanawattano</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Anan</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Bhidayasiri</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Tremor&#x2019;s glove-an innovative electrical muscle stimulation therapy for intractable tremor in Parkinson&#x2019;s disease: a randomized sham-controlled trial</article-title>. <source>J. Neurological Sci.</source> <volume>381</volume>, <fpage>331</fpage>&#x2013;<lpage>340</lpage>. <pub-id pub-id-type="doi">10.1016/j.jns.2017.08.3246</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kalman</surname>
<given-names>R. E.</given-names>
</name>
</person-group> (<year>1960</year>). <article-title>A new approach to linear filtering and prediction problems</article-title>, <source>J. Basic Eng</source>, <volume>82</volume>, <fpage>35</fpage>, <lpage>45</lpage>. <pub-id pub-id-type="doi">10.1115/1.3662552</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaufmann</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Bauersfeld</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Loquercio</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Koltun</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Scaramuzza</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Champion-level drone racing using deep reinforcement learning</article-title>. <source>Nature</source> <volume>620</volume>, <fpage>982</fpage>&#x2013;<lpage>987</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-023-06419-4</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ketteringham</surname>
<given-names>L. P.</given-names>
</name>
<name>
<surname>Western</surname>
<given-names>D. G.</given-names>
</name>
<name>
<surname>Neild</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Hyde</surname>
<given-names>R. A.</given-names>
</name>
<name>
<surname>Jones</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Davies-Smith</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Inverse dynamics modelling of upper-limb tremor, with cross-correlation analysis</article-title>. <source>Healthc. Technol. Lett.</source> <volume>1</volume>, <fpage>59</fpage>&#x2013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.1049/htl.2013.0030</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lamptey</surname>
<given-names>R. N.</given-names>
</name>
<name>
<surname>Chaulagain</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Trivedi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Gothwal</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Layek</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A review of the common neurodegenerative disorders: current therapeutic approaches and the potential role of nanotherapeutics</article-title>. <source>Int. J. Mol. Sci.</source> <volume>23</volume>, <fpage>1851</fpage>. <pub-id pub-id-type="doi">10.3390/ijms23031851</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lang</surname>
<given-names>A. E.</given-names>
</name>
<name>
<surname>Lozano</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>1998a</year>). <article-title>Medical progress: Parkinson&#x2019;s disease (first of two parts)</article-title>. <source>N. Engl. J. Med.</source> <volume>339</volume>, <fpage>1044</fpage>&#x2013;<lpage>1053</lpage>. <pub-id pub-id-type="doi">10.1056/nejm199810083391506</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lang</surname>
<given-names>A. E.</given-names>
</name>
<name>
<surname>Lozano</surname>
<given-names>A. M.</given-names>
</name>
</person-group> (<year>1998b</year>). <article-title>Parkinson&#x2019;s disease</article-title>. <source>N. Engl. J. Med.</source> <volume>339</volume>, <fpage>1130</fpage>&#x2013;<lpage>1143</lpage>. <pub-id pub-id-type="doi">10.1056/nejm199810153391607</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lora-Millan</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Delgado-Oleas</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Benito-Le&#xf3;n</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Rocon</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A review on wearable technologies for tremor suppression</article-title>. <source>Front. neurology</source> <volume>12</volume>, <fpage>700600</fpage>. <pub-id pub-id-type="doi">10.3389/fneur.2021.700600</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Louis</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>Marder</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Cote</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Pullman</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ford</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wilder</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>1995</year>). <article-title>Differences in the prevalence of essential tremor among elderly african Americans, whites, and hispanics in northern manhattan, NY</article-title>. <source>Archives Neurology</source> <volume>52</volume>, <fpage>1201</fpage>&#x2013;<lpage>1205</lpage>. <pub-id pub-id-type="doi">10.1001/archneur.1995.00540360079019</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Androwis</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Adamovich</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Nunez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Robust walking control of a lower limb rehabilitation exoskeleton coupled with a musculoskeletal model via deep reinforcement learning</article-title>. <source>J. neuroengineering rehabilitation</source> <volume>20</volume>, <fpage>34</fpage>&#x2013;<lpage>19</lpage>. <pub-id pub-id-type="doi">10.1186/s12984-023-01147-2</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Androwis</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Adamovich</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Nunez</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Zhou</surname>
<given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Reinforcement learning and control of a lower extremity exoskeleton for squat assistance</article-title>. <source>Front. Robotics AI</source> <volume>8</volume>, <fpage>702845</fpage>. <pub-id pub-id-type="doi">10.3389/frobt.2021.702845</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Luo</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Jiang</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Dominguez Silva</surname>
<given-names>I.</given-names>
</name>
<etal/>
</person-group> (<year>2024</year>). <article-title>Experiment-free exoskeleton assistance via learning in simulation</article-title>. <source>Nature</source> <volume>630</volume>, <fpage>353</fpage>&#x2013;<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-024-07382-4</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mnih</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kavukcuoglu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Antonoglou</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Wierstra</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>Playing atari with deep reinforcement learning</article-title>. <comment>arXiv preprint arXiv:1312.5602</comment>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oliveira</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Coelho</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Carvalho</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Ferreira-Pinto</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Vaz</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Aguiar</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Machine learning for adaptive deep brain stimulation in Parkinson&#x2019;s disease: closing the loop</article-title>. <source>J. Neurology</source> <volume>270</volume>, <fpage>5313</fpage>&#x2013;<lpage>5326</lpage>. <pub-id pub-id-type="doi">10.1007/s00415-023-11873-1</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Otis</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Warren</surname>
<given-names>R. F.</given-names>
</name>
<name>
<surname>Backus</surname>
<given-names>S. I.</given-names>
</name>
<name>
<surname>Santner</surname>
<given-names>T. J.</given-names>
</name>
<name>
<surname>Mabrey</surname>
<given-names>J. D.</given-names>
</name>
</person-group> (<year>1990</year>). <article-title>Torque production in the shoulder of the normal young adult male: the interaction of function, dominance, joint angle, and angular velocity</article-title>. <source>Am. J. sports Med.</source> <volume>18</volume>, <fpage>119</fpage>&#x2013;<lpage>123</lpage>. <pub-id pub-id-type="doi">10.1177/036354659001800201</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pascual-Valdunciel</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gonz&#xe1;lez-S&#xe1;nchez</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Muceli</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ad&#xe1;n-Barrientos</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Escobar-Segura</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>P&#xe9;rez-S&#xe1;nchez</surname>
<given-names>J. R.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>Intramuscular stimulation of muscle afferents attains prolonged tremor reduction in essential tremor patients</article-title>. <source>IEEE Trans. Biomed. Eng.</source> <volume>68</volume>, <fpage>1768</fpage>&#x2013;<lpage>1776</lpage>. <pub-id pub-id-type="doi">10.1109/tbme.2020.3015572</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>X. B.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Van de Panne</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2018a</year>). <article-title>Deepmimic: example-guided deep reinforcement learning of physics-based character skills</article-title>. <source>ACM Trans. Graph. (TOG)</source> <volume>37</volume>, <fpage>1</fpage>&#x2013;<lpage>14</lpage>. <pub-id pub-id-type="doi">10.1145/3197517.3201311</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Peng</surname>
<given-names>X. B.</given-names>
</name>
<name>
<surname>Andrychowicz</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Zaremba</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2018b</year>). &#x201c;<article-title>Sim-to-real transfer of robotic control with dynamics randomization</article-title>,&#x201d; in <conf-name>2018 IEEE International Conference on Robotics and Automation (ICRA) (IEEE)</conf-name>, <conf-loc>USA</conf-loc>, <conf-date>May 1 &#x2014; Fri May 5</conf-date>, <fpage>3803</fpage>&#x2013;<lpage>3810</lpage>. <pub-id pub-id-type="doi">10.1109/icra.2018.8460528</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Plagenhoef</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Evans</surname>
<given-names>F. G.</given-names>
</name>
<name>
<surname>Abdelnour</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>1983</year>). <article-title>Anatomical data for analyzing human motion</article-title>. <source>Res. Q. Exerc. sport</source> <volume>54</volume>, <fpage>169</fpage>&#x2013;<lpage>178</lpage>. <pub-id pub-id-type="doi">10.1080/02701367.1983.10605290</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Riviere</surname>
<given-names>C. N.</given-names>
</name>
<name>
<surname>Reich</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>Thakor</surname>
<given-names>N. V.</given-names>
</name>
</person-group> (<year>1997</year>). <article-title>Adaptive fourier modeling for quantification of tremor</article-title>. <source>J. Neurosci. methods</source> <volume>74</volume>, <fpage>77</fpage>&#x2013;<lpage>87</lpage>. <pub-id pub-id-type="doi">10.1016/s0165-0270(97)02263-2</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rocon</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Belda-Lois</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Ruiz</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Manto</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Moreno</surname>
<given-names>J. C.</given-names>
</name>
<name>
<surname>Pons</surname>
<given-names>J. L.</given-names>
</name>
</person-group> (<year>2007a</year>). <article-title>Design and validation of a rehabilitation robotic exoskeleton for tremor assessment and suppression</article-title>. <source>IEEE Trans. neural Syst. rehabilitation Eng.</source> <volume>15</volume>, <fpage>367</fpage>&#x2013;<lpage>378</lpage>. <pub-id pub-id-type="doi">10.1109/tnsre.2007.903917</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rocon</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Manto</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pons</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Camut</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Belda</surname>
<given-names>J. M.</given-names>
</name>
</person-group> (<year>2007b</year>). <article-title>Mechanical suppression of essential tremor</article-title>. <source>Cerebellum</source> <volume>6</volume>, <fpage>73</fpage>&#x2013;<lpage>78</lpage>. <pub-id pub-id-type="doi">10.1080/14734220601103037</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Sadeghi</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Levine</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Cad2rl: real single-image flight without a single real image</article-title>. <source>arXiv preprint arXiv:1611.04201</source>.</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schulman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Wolski</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Dhariwal</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Radford</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Klimov</surname>
<given-names>O.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Proximal policy optimization algorithms</article-title>. <comment>
<italic>arXiv preprint arXiv:1707.06347</italic>
</comment>
</citation>
</ref>
<ref id="B53">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lever</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Heess</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Degris</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Wierstra</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Riedmiller</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). &#x201c;<article-title>Deterministic policy gradient algorithms</article-title>,&#x201d; in <conf-name>International conference on machine learning</conf-name> <conf-loc>USA</conf-loc>, <conf-date>19 Jul, 2025</conf-date>, (<publisher-name>Pmlr</publisher-name>), <fpage>387</fpage>&#x2013;<lpage>395</lpage>.</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Siviy</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Baker</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>Quinlivan</surname>
<given-names>B. T.</given-names>
</name>
<name>
<surname>Porciuncula</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Swaminathan</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Awad</surname>
<given-names>L. N.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Opportunities and challenges in the development of exoskeletons for locomotor assistance</article-title>. <source>Nat. Biomed. Eng.</source> <volume>7</volume>, <fpage>456</fpage>&#x2013;<lpage>472</lpage>. <pub-id pub-id-type="doi">10.1038/s41551-022-00984-1</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Skaramagkas</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Andrikopoulos</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Manesis</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>An experimental investigation of essential hand tremor suppression via a soft exoskeletal glove</article-title>,&#x201d; in <conf-name>2020 European control conference (ECC)</conf-name> <conf-loc>China</conf-loc>, <conf-date>12 May 2020</conf-date>, (<publisher-name>IEEE</publisher-name>), <fpage>889</fpage>&#x2013;<lpage>894</lpage>.</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Slade</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Kochenderfer</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Delp</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Collins</surname>
<given-names>S. H.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Personalizing exoskeleton assistance while walking in the real world</article-title>. <source>Nature</source> <volume>610</volume>, <fpage>277</fpage>&#x2013;<lpage>282</lpage>. <pub-id pub-id-type="doi">10.1038/s41586-022-05191-1</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sutton</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Barto</surname>
<given-names>A. G.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Reinforcement learning: an introduction</source>. <publisher-loc>China</publisher-loc>, <publisher-name>MIT press</publisher-name>.</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taheri</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Case</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Richer</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>2013</year>). <article-title>Robust controller for tremor suppression at musculoskeletal level in human wrist</article-title>. <source>IEEE Trans. neural Syst. rehabilitation Eng.</source> <volume>22</volume>, <fpage>379</fpage>&#x2013;<lpage>388</lpage>. <pub-id pub-id-type="doi">10.1109/tnsre.2013.2295034</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tobin</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Fong</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Ray</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Schneider</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Zaremba</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Abbeel</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Domain randomization for transferring deep neural networks from simulation to the real world</article-title>, <source>2017 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)</source>, <fpage>23</fpage>, <lpage>30</lpage>. <pub-id pub-id-type="doi">10.1109/iros.2017.8202133</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Hasselt</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Guez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Deep reinforcement learning with double q-learning</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>30</volume>. <pub-id pub-id-type="doi">10.1609/aaai.v30i1.10295</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Gao</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Jamming enabled variable stiffness wrist exoskeleton for tremor suppression</article-title>. <source>IEEE Robotics Automation Lett.</source> <volume>8</volume>, <fpage>3693</fpage>&#x2013;<lpage>3700</lpage>. <pub-id pub-id-type="doi">10.1109/lra.2023.3270747</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Welch</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Bishop</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1995</year>). <article-title>An introduction to the kalman filter</article-title>. <publisher-loc>Chapel Hill, NC, USA</publisher-loc>: <publisher-name>University of North Carolina at Chapel Hill</publisher-name>.</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>She</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Z.-T.</given-names>
</name>
<name>
<surname>Yang</surname>
<given-names>Z.</given-names>
</name>
<name>
<surname>Sato</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A tremor-suppressing strategy based on the equivalent-input-disturbance approach</article-title>. <source>IEEE/ASME Trans. Mechatronics</source> <volume>29</volume>, <fpage>3971</fpage>&#x2013;<lpage>3980</lpage>. <pub-id pub-id-type="doi">10.1109/tmech.2024.3375911</pub-id>
</citation>
</ref>
<ref id="B64">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zahedi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Tan</surname>
<given-names>U.-X.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>A novel exoskeleton system based on magnetorheological fluid for tremor suppression of wrist joints</article-title>,&#x201d; in <conf-name>2019 IEEE 16th international conference on rehabilitation robotics (ICORR)</conf-name> <conf-loc>USA</conf-loc>, <conf-date>24-28 June 2019</conf-date>, (<publisher-name>IEEE</publisher-name>), <fpage>1115</fpage>&#x2013;<lpage>1120</lpage>.</citation>
</ref>
<ref id="B65">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zahedi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Yi</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Zhang</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A soft exoskeleton for tremor suppression equipped with flexible semiactive actuator</article-title>. <source>Soft Robot.</source> <volume>8</volume>, <fpage>432</fpage>&#x2013;<lpage>447</lpage>. <pub-id pub-id-type="doi">10.1089/soro.2019.0194</pub-id>
</citation>
</ref>
<ref id="B66">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Naish</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Jenkins</surname>
<given-names>M. E.</given-names>
</name>
<name>
<surname>Trejos</surname>
<given-names>A. L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Design and validation of a novel mechatronic transmission system for a wearable tremor suppression device</article-title>. <source>Robotics Aut. Syst.</source> <volume>91</volume>, <fpage>38</fpage>&#x2013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.1016/j.robot.2016.12.009</pub-id>
</citation>
</ref>
<ref id="B67">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zwerus</surname>
<given-names>E. L.</given-names>
</name>
<name>
<surname>Willigenburg</surname>
<given-names>N. W.</given-names>
</name>
<name>
<surname>Scholtes</surname>
<given-names>V. A.</given-names>
</name>
<name>
<surname>Somford</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>Eygendaal</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>van den Bekerom</surname>
<given-names>M. P.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Normative values and affecting factors for the elbow range of motion</article-title>. <source>Shoulder and Elb.</source> <volume>11</volume>, <fpage>215</fpage>&#x2013;<lpage>224</lpage>. <pub-id pub-id-type="doi">10.1177/1758573217728711</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>
