<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Pharmacol.</journal-id>
<journal-title>Frontiers in Pharmacology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Pharmacol.</abbrev-journal-title>
<issn pub-type="epub">1663-9812</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1270443</article-id>
<article-id pub-id-type="doi">10.3389/fphar.2023.1270443</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Pharmacology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Simulating clinical trials for model-informed precision dosing: using warfarin treatment as a use case</article-title>
<alt-title alt-title-type="left-running-head">Augustin et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fphar.2023.1270443">10.3389/fphar.2023.1270443</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Augustin</surname>
<given-names>David</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2034785/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Lambert</surname>
<given-names>Ben</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Robinson</surname>
<given-names>Martin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wang</surname>
<given-names>Ken</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/503334/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Gavaghan</surname>
<given-names>David</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/494247/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Computer Science</institution>, <institution>University of Oxford</institution>, <addr-line>Oxford</addr-line>, <country>United Kingdom</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>College of Engineering</institution>, <institution>Mathematics and Physical Sciences</institution>, <institution>University of Exeter</institution>, <addr-line>Exeter</addr-line>, <country>United Kingdom</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Research and Early Development</institution>, <institution>F. Hoffmann-La Roche AG</institution>, <addr-line>Basel</addr-line>, <country>Switzerland</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1907788/overview">Fawzy Elbarbry</ext-link>, Pacific University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/780288/overview">Thomas Polasek</ext-link>, Monash University, Australia</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2066255/overview">Wilhelm Huisinga</ext-link>, University of Potsdam, Germany</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: David Augustin, <email>david.augustin@cs.ox.ac.uk</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>10</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>14</volume>
<elocation-id>1270443</elocation-id>
<history>
<date date-type="received">
<day>02</day>
<month>08</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>05</day>
<month>10</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Augustin, Lambert, Robinson, Wang and Gavaghan.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Augustin, Lambert, Robinson, Wang and Gavaghan</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Treatment response variability across patients is a common phenomenon in clinical practice. For many drugs this inter-individual variability does not require much (if any) individualisation of dosing strategies. However, for some drugs, including chemotherapies and some monoclonal antibody treatments, individualisation of dosages are needed to avoid harmful adverse events. Model-informed precision dosing (MIPD) is an emerging approach to guide the individualisation of dosing regimens of otherwise difficult-to-administer drugs. Several MIPD approaches have been suggested to predict dosing strategies, including regression, reinforcement learning (RL) and pharmacokinetic and pharmacodynamic (PKPD) modelling. A unified framework to study the strengths and limitations of these approaches is missing. We develop a framework to simulate clinical MIPD trials, providing a cost and time efficient way to test different MIPD approaches. Central for our framework is a clinical trial model that emulates the complexities in clinical practice that challenge successful treatment individualisation. We demonstrate this framework using warfarin treatment as a use case and investigate three popular MIPD methods: 1. Neural network regression; 2. Deep RL; and 3. PKPD modelling. We find that the PKPD model individualises warfarin dosing regimens with the highest success rate and the highest efficiency: 75.1% of the individuals display INRs inside the therapeutic range at the end of the simulated trial; and the median time in the therapeutic range (TTR) is 74%. In comparison, the regression model and the deep RL model have success rates of 47.0% and 65.8%, and median TTRs of 45% and 68%. We also find that the MIPD models can attain different degrees of individualisation: the Regression model individualises dosing regimens up to variability explained by covariates; the Deep RL model and the PKPD model individualise dosing regimens accounting also for additional variation using monitoring data. However, the Deep RL model focusses on control of the treatment response, while the PKPD model uses the data also to further the individualisation of predictions.</p>
</abstract>
<kwd-group>
<kwd>MIPD</kwd>
<kwd>precision dosing</kwd>
<kwd>clinical trial simulation</kwd>
<kwd>inter-individual variability</kwd>
<kwd>PKPD modelling</kwd>
<kwd>reinforcement learning</kwd>
<kwd>deep learning</kwd>
<kwd>warfarin</kwd>
</kwd-group>
<contract-sponsor id="cn001">Engineering and Physical Sciences Research Council<named-content content-type="fundref-id">10.13039/501100000266</named-content>
</contract-sponsor>
<contract-sponsor id="cn002">Biotechnology and Biological Sciences Research Council<named-content content-type="fundref-id">10.13039/501100000268</named-content>
</contract-sponsor>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Drug Metabolism and Transport</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Model-informed precision dosing (MIPD) is an emerging technique used to individualise dosing regimens of otherwise difficult-to-administer drugs (<xref ref-type="bibr" rid="B49">Sheiner, 1969</xref>; <xref ref-type="bibr" rid="B30">Keizer et al., 2018</xref>; <xref ref-type="bibr" rid="B11">Darwich et al., 2021</xref>). Typical examples that would benefit from MIPD are drugs with narrow therapeutic windows and large treatment response variability, such as warfarin, docetaxel and infliximab (<xref ref-type="bibr" rid="B56">Wadelius and Pirmohamed, 2007</xref>; <xref ref-type="bibr" rid="B16">Gill et al., 2016</xref>; <xref ref-type="bibr" rid="B35">Ma et al., 2021</xref>). Other examples include antibiotics, like vancomycin, where MIPD has been suggested and partially implemented to guide dosing strategies for critically ill patients, balancing the treatment of severe infections and the risks for harmful adverse events (<xref ref-type="bibr" rid="B8">Broeker et al., 2019</xref>; <xref ref-type="bibr" rid="B59">Wicha et al., 2021</xref>; <xref ref-type="bibr" rid="B39">Matsumoto et al., 2022</xref>). MIPD may also be used to efficiently adapt dosing regimens to continuously changing conditions, for example to stabilise the blood glucose levels of diabetes patients with insulin treatment (<xref ref-type="bibr" rid="B58">Wang et al., 2019</xref>; <xref ref-type="bibr" rid="B62">Zhu et al., 2020</xref>).</p>
<p>The most prominent MIPD methods are regression, reinforcement learning (RL) and pharmacokinetic and pharmacodynamic (PKPD) modelling (<xref ref-type="bibr" rid="B28">Johnson et al., 2011</xref>; <xref ref-type="bibr" rid="B27">Johnson et al., 2017</xref>; <xref ref-type="bibr" rid="B30">Keizer et al., 2018</xref>; <xref ref-type="bibr" rid="B47">Ribba et al., 2020</xref>). These methods may be further categorised into different subvariants. RL variants include, for example, model-free algorithms such as Q learning (<xref ref-type="bibr" rid="B61">Zadeh et al., 2023</xref>) and model-based algorithms such as Monte Carlo tree search (<xref ref-type="bibr" rid="B38">Maier et al., 2021</xref>). MIPD variants of PKPD modelling include maximum a posteriori-guided dosing and Bayesian data assimilation-guided dosing (<xref ref-type="bibr" rid="B37">Maier et al., 2020</xref>). Other variants include the modelling of virtual twins using physiology-based PK (PBPK) and quantitative systems pharmacology (QSP) models (<xref ref-type="bibr" rid="B45">Polasek and Rostami-Hodjegan, 2020</xref>). Although differences across MIPD methods exist, all have in common that they need to process data in two steps in order to make individualised predictions. First models are fitted to population data. This step establishes a relationship between patient characteristics and dosing strategies. In a second step, data specific to the to-be-treated patient is used to predict individualised dosing regimens. For example, regression models have been fitted using records of genetic information and dosages across patients (<xref ref-type="bibr" rid="B14">Gage et al., 2008</xref>; <xref ref-type="bibr" rid="B33">Klein et al., 2009</xref>; <xref ref-type="bibr" rid="B17">Gong et al., 2011</xref>), enabling the prediction of individualised dosages based on genetic information.</p>
<p>The data used for model fitting and dosing regimen individualisation differ substantially across MIPD methods and include clinical factors, genetic factors and monitoring data. The type and volume of data are key to both the accuracy of predictions and the ease of implementation in clinical practice (<xref ref-type="bibr" rid="B10">Darwich et al., 2017</xref>; <xref ref-type="bibr" rid="B46">Ribba et al., 2022</xref>). The more data are collected, the better dosing regimens can be individualised. However, practical constraints limit how much and what type of data may be available for MIPD. As a result, the trade-off between accuracy and practicality needs to be considered when applying MIPD approaches to medicines. The systematic study of this trade-off for a specific application is, however, complicated by the astronomical costs of clinical trials, rendering repeated clinical trials for different MIPD methods infeasible. In this study, we propose a framework for the simulation of clinical MIPD trials, facilitating a resource-efficient way to test and develop MIPD approaches.</p>
<p>Using simulations to understand MIPD approaches is not a new concept and several MIPD simulation studies exist in the literature. <xref ref-type="bibr" rid="B42">Moore et al. (2004)</xref> and <xref ref-type="bibr" rid="B46">Ribba et al. (2022)</xref> use simulated treatment responses to investigate RL approaches as a strategy to individualise dosages of anaesthetics in intensive care units. For the simulations, they used a semi-mechanistic PKPD model to emulate the time course of treatment responses and a nonlinear mixed effects (NLME) model structure to capture inter-individual variability (IIV). A similar approach is adopted by <xref ref-type="bibr" rid="B37">Maier et al. (2020</xref>, <xref ref-type="bibr" rid="B38">2021)</xref> to study the individualisation of paclitaxel-based chemotherapy using different MIPD approaches, including PKPD modelling, RL and a hybrid PKPD-RL approach. An NLME model simulation approach is also used by <xref ref-type="bibr" rid="B61">Zadeh et al. (2023)</xref> to study deep RL as a candidate for MIPD of warfarin. <xref ref-type="bibr" rid="B1">Abrantes et al. (2019)</xref> extend this NLME simulation approach, adding inter-occasion variability (IOV) of treatment responses as an extra dimension to the MIPD simulation. They model IOV following <xref ref-type="bibr" rid="B29">Karlsson and Sheiner (1993)</xref> and randomly vary the PKPD model parameters of virtual patients over time. An analogous approach is used by <xref ref-type="bibr" rid="B31">Keutzer and Simonsson (2020)</xref> to understand MIPD-based treatment individualisation of rifampicin.</p>
<p>We propose an extended framework for the simulation of clinical MIPD trials. Our framework complements the emulation of IIV and IOV by other previously established elements of clinical trial simulation (<xref ref-type="bibr" rid="B25">Holford et al., 2000</xref>; <xref ref-type="bibr" rid="B26">Holford et al., 2010</xref>). In particular, our framework includes treatment response emergence from complex interactions of pharmacological and physiological processes as a central feature of the trial simulation. In addition, deviations of dose administrations from nominal dosing regimens, as well as delayed monitoring measurements are incorporated in the simulation to more faithfully represent practical limitations of monitoring-based MIPD approaches.</p>
<p>The article is divided into three sections: methods; results and discussion; and conclusion. In the methods, we first introduce the general framework for MIPD trial simulation and subsequently demonstrate its implementation in terms of a clinical trial model for the warfarin use case. We then introduce the three MIPD models, which we will investigate with the help of the clinical trial model. The considered models are: 1. a neural network regression model; 2. a deep reinforcement learning model; and 3. a PKPD model. In the results and discussion, we use the clinical trial model to simulate MIPD trials for each of the three models and analyse the strengths and limitations of the MIPD models. In our analysis, we pay careful attention to attributing generic strengths and limitations to the MIPD methodology and specific strengths and limitations to our implementations of the models. In the conclusion, we summarise our findings and propose future directions for MIPD research.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<p>We first introduce the proposed framework for clinical MIPD trial simulation and discuss the specific implementation for warfarin. We then outline the investigated MIPD methods. The data, models and scripts used in this article are hosted on GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/DavAug/mipd-warfarin">https://github.com/DavAug/mipd-warfarin</ext-link>). A user-friendly API for MIPD approaches using PKPD modelling has been implemented in the open source Python package chi (<xref ref-type="bibr" rid="B3">Augustin, 2021</xref>). MIPD approaches using neural networks were implemented in PyTorch (<xref ref-type="bibr" rid="B43">Paszke et al., 2019</xref>).</p>
<sec id="s2-1">
<title>2.1 General framework for MIPD trial simulation</title>
<p>The objective of MIPD is to achieve desired treatment outcomes across individuals despite large treatment response variability by optimising individual dosing regimens. Challenges for this individualisation are nonlinear and delayed treatment responses (<xref ref-type="bibr" rid="B36">Mager, 2006</xref>; <xref ref-type="bibr" rid="B12">Dirks and Meibohm, 2010</xref>; <xref ref-type="bibr" rid="B54">V&#xe9;ronneau-Veilleux et al., 2020</xref>), making it difficult to adjust and extrapolate dosages based on feedback from monitoring. Variability of the treatment response as a result of time-varying pharmacokinetics, concomitant food-intake, comedication and disease progression further complicate the dosing regimen adjustment (<xref ref-type="bibr" rid="B31">Keutzer and Simonsson, 2020</xref>). A realistic assessment of MIPD approaches in a simulated trial therefore needs to account not only for IIV, but also for treatment response delays, nonlinearities and IOV. However, PKPD-related aspects are not the only factors influencing the success of MIPD methods. There are also practical limitations for MIPD, including imperfect measurements and deviations from nominal dosing and monitoring schedules (<xref ref-type="bibr" rid="B25">Holford et al., 2000</xref>; <xref ref-type="bibr" rid="B26">Holford et al., 2010</xref>). While measurement noise is commonly included in simulations, the variability in the execution of the trial often remains neglected. Deviations from the nominal schedule can impact the success of MIPD methods, especially when they are not fed back into the model.</p>
<p>To faithfully emulate the performance of MIPD in clinical practice, our simulation framework accounts for these PKPD-related and practical challenges using a clinical trial (CT) model composed of five components: 1. a mechanistic model; 2. a population model; 3. an inter-occasion model; 4. an execution model and 5. a measurement model (see left panel in <xref ref-type="fig" rid="F1">Figure 1</xref>). We describe each of these components in <xref ref-type="sec" rid="s2-1-2">Section 2.1.2</xref>. The right panel of the figure illustrates the workflow of using the CT model for simulating MIPD trials.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Framework for MIPD trial simulation. The left panel shows the components of the clinical trial (CT) model: 1. A mechanistic model; 2. A population model; 3. An inter-occasion model; 4. An execution model; and 5. A measurement model. The right panel shows the workflow leading up to the MIPD trial simulation. First, the CT model is used to simulate typical clinical trial data, available prior to the MIPD trial. Second, the simulated trial data is used to fit the parameters of the MIPD model. This emulates the starting point of MIPD trials in practice. Finally, the CT model and the fitted MIPD model are used to simulate the MIPD trial.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g001.tif"/>
</fig>
<sec id="s2-1-1">
<title>2.1.1 Workflow of MIPD trial simulation</title>
<p>Our method of MIPD trial simulation involves three steps (see right panel in <xref ref-type="fig" rid="F1">Figure 1</xref>): 1. simulation of pre-MIPD clinical trial data; 2. fitting of the MIPD model to this simulated data; 3. simulation of the MIPD trial. The first two steps emulate the MIPD model development that takes place in practice prior to MIPD trials. In our method, we first simulate typical clinical data from e.g. phase I trials, phase II trials and/or phase III trials using the CT model. We then fit the MIPD model parameters to the simulated data. The details of the fitting process are specific to the MIPD model and are presented in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>. It is important that the simulated data are used for the fitting, even if real clinical data are available, in order to facilitate a clear attribution of limitations observed in a simulated MIPD trial to the MIPD model. If instead, the MIPD model was fitted to real clinical data, the approximation error of the CT model with respect to the data-generating process of the real clinical trial may also contribute to limitations observed in the simulated trials, making it harder to draw conclusions about the MIPD model. The real data should, instead, be used to calibrate the CT model prior to the data simulation in order to minimise the approximation error as much as possible. The final step of the workflow is the simulation of the MIPD trial using the fitted MIPD model and the CT model.</p>
</sec>
<sec id="s2-1-2">
<title>2.1.2 Components of the CT model</title>
<p>We now describe the components of the CT model. By design, the components of the simulation are non-overlapping and are as modular as possible. This simplifies the model structure and enables an iterative development of CT models, making it possible to replace or further develop individual components without requiring changes in other model components.</p>
<p>1. The mechanistic model: This component models the dynamics of treatment responses as a function of time, <italic>t</italic>, and the dosing regimen, <italic>r</italic>,<disp-formula id="e1">
<mml:math id="m1">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(1)</label>
</disp-formula>
<inline-formula id="inf1">
<mml:math id="m2">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> denotes quantities of interest that may be monitored in clinical practice, and <italic>&#x3c8;</italic> denotes the parameters of the model. The main purpose of the mechanistic model is to faithfully reflect nonlinearities and delays of the treatment response, emergent from complex cascades of pharmacological and physiological processes. Emulating this complexity provides a tool to test the ability of different MIPD approaches to approximate the treatment response and predict individualised dosing regimens. Popular choices to simulate treatment response dynamics include PKPD models and quantitative systems pharmacology (QSP) models (<xref ref-type="bibr" rid="B47">Ribba et al., 2020</xref>; <xref ref-type="bibr" rid="B6">Azer et al., 2021</xref>; <xref ref-type="bibr" rid="B38">Maier et al., 2021</xref>).</p>
<p>An example mechanistic model of warfarin treatment developed by <xref ref-type="bibr" rid="B57">Wajima et al. (2009)</xref> is illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>. Warfarin is an oral anticoagulant widely used for the prevention and treatment of venous thrombosis, pulmonary embolism and thromboembolic complications associated with atrial fibrillation and/or cardiac valve replacement (<xref ref-type="bibr" rid="B13">FDA, 2010</xref>). The left panel shows the 53 blood components described by the model, including warfarin, vitamin K and different coagulation factors, such as thrombin and fibrin. Edges between the components represent interactions, such as transitions, inhibitions or activations. The monitored quantity of the treatment response, <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, is the prothrombin time. The prothrombin time measures the time it takes for plasma to clot after exposure to a thromboplastin reagent and is routinely measured in clinical practice. In Wajima et al.&#x2019;s model, this prothrombin time test is simulated by measuring the time elapsed between adding the reagent (300&#xa0;nM tissue factor) and reaching a fibrin area-under-the-curve (AUC) of 1500&#xa0;nMs (see right panel in <xref ref-type="fig" rid="F2">Figure 2</xref>). The prothrombin time is commonly reported in terms of the international normalised ratio (INR), which measures the prothrombin time of a patient&#x2019;s blood sample in units of the prothrombin time of a reference sample. We will use this model in our warfarin clinical trial simulation (see <xref ref-type="sec" rid="s2-2">Section 2.2</xref> for details).</p>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Warfarin clinical trial simulation - the mechanistic model. The figure shows Wajima et al.&#x2019;s model of the warfarin treatment response mechanism. The left panel shows the model. Nodes represent states of the model, including warfarin (red), vitamin K (green) and different coagulation factors (blue). Central coagulation factors include thrombin (light blue) and fibrin (yellow). Multiple nodes for warfarin refer to the drug amount in different compartments, while multiple nodes for vitamin K also refer to different forms of vitamin K, such as vitamin K epoxide and vitamin K hydroquinone. Interactions between states, such as transitions, inhibitions or activations, are represented by edges. The right panel shows the model simulation of the prothrombin time test. The arrow indicates the prothrombin time which marks the time elapsed between exposure to tissue factor and reaching a fibrin AUC of 1500&#xa0;nMs.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g002.tif"/>
</fig>
<p>2. The population model: This component models the variability in the treatment response across individuals using a mixed effects model extension of the mechanistic model. A mixed effects model defines a population distribution of model parameters<disp-formula id="e2">
<mml:math id="m4">
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(2)</label>
</disp-formula>capturing the differences between individuals, i.e., the IIV (<xref ref-type="bibr" rid="B34">Lavielle, 2014</xref>; <xref ref-type="bibr" rid="B4">Augustin et al., 2023</xref>). <italic>&#x3b8;</italic> denotes the parameters of the population distribution. Each sample, <italic>&#x3c8;</italic>, from the population distribution represents an individual with treatment response <inline-formula id="inf3">
<mml:math id="m5">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. Thus, differences between individuals arise in this model structure from differences in the mechanistic model parameters. For some applications, these differences can be partially explained by covariates, <italic>&#x3c7;</italic>, which may divide the population into subpopulations, <italic>p</italic>(<italic>&#x3c8;</italic>&#x7c;<italic>&#x3b8;</italic>, <italic>&#x3c7;</italic>). The full population distribution across covariates is then given by the average of the supopulations weighted by the relative frequency of the covariates, <inline-formula id="inf4">
<mml:math id="m6">
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="double-struck">E</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c7;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="[" close="]">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c7;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula>.</p>
<p>Covariates of the variability can range from clinical factors to genetic factors. For example for warfarin treatment, the VKORC1 genotype explains 27% of the observed response variability (<xref ref-type="bibr" rid="B55">Wadelius et al., 2009</xref>). Other covariates of warfarin treatment include mutations in the CYP2C9 gene and the age of the patient. The link between covariates and pharmacological or physiological processes makes it possible to define mixed effects models that reflect the mechanistic relationship between covariates and the treatment response variability (<xref ref-type="bibr" rid="B18">Hamberg et al., 2007</xref>; <xref ref-type="bibr" rid="B20">Hamberg et al., 2010</xref>; <xref ref-type="bibr" rid="B23">Hartmann et al., 2016</xref>; <xref ref-type="bibr" rid="B22">Hartmann et al., 2020</xref>). For example, warfarin&#x2019;s mode of action is the inhibition of the vitamin K epoxide reductase complex (VKORC), and mutations in VKORC&#x2019;s subunit 1 (VKORC1) affect the inhibitory activity, which can be implemented with a reduced EC50 parameter in Wajima et al.&#x2019;s mechanistic model.</p>
<p>The left panel of <xref ref-type="fig" rid="F3">Figure 3</xref> illustrates the emergence of inter-individual treatment response variability in our warfarin clinical trial model. The figure shows simulated INR treatment responses of 6 individuals to daily administrations of warfarin. 3 individuals have the GG genotype (VKORC1), the &#x2a;1/&#x2a;1 genotype (CYP2C9) and are 71&#xa0;years old (see blue lines). The remaining 3 individuals have the GA genotype (VKORC1), the &#x2a;1/&#x2a;2 genotype (CYP2C9) and are 46&#xa0;years old (see red lines). The dashed line indicates the treatment response of an average 71&#xa0;year old individual with the GG and &#x2a;1/&#x2a;1 genotypes. We can see that individuals with <italic>&#x3c7;</italic> &#x3d; (GG, &#x2a;1/&#x2a;1, 71) tend to respond less strongly to warfarin treatment than individuals with <italic>&#x3c7;</italic> &#x3d; (GA, &#x2a;1/&#x2a;2, 46). However, there remains substantial IIV that is not explained by covariates.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Warfarin clinical trial simulation&#x2013;sources of treatment response variability. The figure shows each contribution to the treatment response variability in isolation. Panel 1: Illustrates the effect of the population model on the clinical trial simulation. The solid lines indicate the treatment responses of 6 simulated individuals to daily warfarin administrations: 3 of which have the covariates <italic>&#x3c7;</italic> &#x3d; (GG, &#x2a; 1/&#x2a; 1, 71) (see blue lines); and the remaining 3 have the covariates <italic>&#x3c7;</italic> &#x3d; (GA, &#x2a; 1/&#x2a; 2, 46) (see red lines). The typical treatment response across individuals is indicated by a dashed line. The administration times are indicated by blue arrows. Panel 2: Illustrates the effect of the inter-occasion model on the clinical trial simulation. The dashed line shows the treatment response of an individual with a constant vitamin K input rate, i.e., no IOV. The solid lines show the treatment response of the same individual with vitamin K input rates that randomly vary by 10% (blue), 20% (red) and 30% (grey) between days. Panel 3: Illustrates the effect of the execution model on the clinical trial simulation. The dashed line indicates the treatment response of an individual associated with the nominal dosing regimen (hollow arrows). The blue line indicates the treatment response of the same individual associated with the delayed dose administrations (blue arrows). Panel 4: Illustrates the effect of the measurement model on the clinical trial simulation. The dashed line shows the simulated treatment response of an individual and the scatter points show the associated monitoring measurement.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g003.tif"/>
</fig>
<p>3. The inter-occasion model: This component models the variability of the treatment response over time using time-varying modifications of the model parameters<disp-formula id="e3">
<mml:math id="m7">
<mml:mi>&#x3c8;</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
<mml:mspace width="0.17em"/>
<mml:mi>&#x3b7;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(3)</label>
</disp-formula>
<italic>&#x3b7;</italic> denotes the alterations of the model parameters, and <italic>&#x3c8;</italic>&#x2032; denotes the new model parameters. The treatment response of an individual with parameters <italic>&#x3c8;</italic> is now given by the mechanistic model simulation using the time-varying parameters, <inline-formula id="inf5">
<mml:math id="m8">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="B29">Karlsson and Sheiner, 1993</xref>). The role of the inter-occasion model is to implement changes of the treatment response that are not accounted for by the mechanistic model. Such changes can be externally driven, e.g., by concomitant food intake or comedication, or of completely unknown origin (<xref ref-type="bibr" rid="B31">Keutzer and Simonsson, 2020</xref>). The dynamics of <italic>&#x3b7;</italic>, <italic>p</italic>(<italic>&#x3b7;</italic>&#x7c;<italic>t</italic>), are a modelling choice.</p>
<p>A source of inter-occasion variability for warfarin is, for example, the time-varying consumption of vitamin K (<xref ref-type="bibr" rid="B60">Xue et al., 2016</xref>), changing the amount of vitamin K available in the blood. Since warfarin&#x2019;s mode of action is to inhibit VKORC &#x2013; a complex converting one form of vitamin K into another, clotting factor-activating form of vitamin K &#x2013; an increased availability of vitamin K can reverse the treatment effects of warfarin. In the 2<sup>nd</sup> panel of <xref ref-type="fig" rid="F3">Figure 3</xref> we illustrate the effects of varying vitamin K consumption on the warfarin treatment response in the clinical trial simulation. The dashed line shows the treatment response simulation with a constant vitamin K input rate parameter, i.e., no variability in the vitamin K consumption. The solid lines show the treatment response simulations with vitamin K input rates that randomly vary by 10% (blue), 20% (red) and 30% (grey) from day to day.</p>
<p>4. The execution model: This component models unintended deviations from nominal dosing regimens and monitoring schedules during the trial. Nominal dosing regimens are defined by a sequence of doses and administration times, <italic>r</italic> &#x3d; {(<italic>d</italic>
<sub>
<italic>j</italic>
</sub>, <italic>t</italic>
<sub>
<italic>j</italic>
</sub>)}, where <italic>d</italic>
<sub>
<italic>j</italic>
</sub> denotes the <italic>j</italic>th dose and <italic>t</italic>
<sub>
<italic>j</italic>
</sub> denotes the associated administration time. Nominal monitoring schedules are similarly defined by a sequence of measurement times. The actual doses and times are modelled in the execution model using random deviations from the nominal schedules<disp-formula id="e4">
<mml:math id="m9">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2192;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>,</mml:mo>
<mml:mspace width="2em"/>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2192;</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
</mml:math>
<label>(4)</label>
</disp-formula>where &#x394;<italic>d</italic> and &#x394;<italic>t</italic> denote the deviations. The treatment response corresponding to the actual dosing regimen, <inline-formula id="inf6">
<mml:math id="m10">
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msubsup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, is simulated using <inline-formula id="inf7">
<mml:math id="m11">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The role of the execution model is to test the robustness of MIPD approaches in clinical practice by emulating the limited control over dose administrations and monitoring. By choosing to report nominal schedules as opposed to actual schedules, the execution model can also be used to mirror common inaccuracies of clinical data. The distributions of dose and time deviations, <italic>p</italic>(&#x394;<italic>d</italic>) and <italic>p</italic>(&#x394;<italic>t</italic>), are modelling choices and may differ between dosing and measurement schedules. While not considered in this article, the execution model may be extended to include missed administrations or missed measurements during the trial. Assuming no persistence, this can be implemented using draws of Bernoulli random variables associated with each time point, indicating whether or not a dose was administered or a measurement was taken (<xref ref-type="bibr" rid="B25">Holford et al., 2000</xref>; <xref ref-type="bibr" rid="B26">Holford et al., 2010</xref>). For infusions, deviations in the duration of the administration may also be modelled.</p>
<p>In the 3<sup>rd</sup> panel of <xref ref-type="fig" rid="F3">Figure 3</xref> we illustrate the effects of delayed dose administrations on the treatment response in the warfarin trial simulation. Doses are administered daily. The nominal dosing regimen is illustrated by hollow arrows. The actual dosing regimen with delayed administrations is illustrated by blue arrows. The associated treatment response simulations are indicated by the dashed line (nominal) and by the solid line (actual).</p>
<p>5. The measurement model: This component models the limited accuracy of treatment response measurements<disp-formula id="e5">
<mml:math id="m12">
<mml:mi>y</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3b5;</mml:mi>
<mml:mo>,</mml:mo>
</mml:math>
<label>(5)</label>
</disp-formula>where <italic>y</italic> denotes the measurement and <italic>&#x25b;</italic> denotes the measurement error. This defines a distribution of measurements around the mechanistic model output, <inline-formula id="inf8">
<mml:math id="m13">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>, at each time point <italic>t</italic>, <italic>p</italic>(<italic>y</italic>&#x7c;<italic>t</italic>, <italic>r</italic>&#x2032;, <italic>&#x3c8;</italic>&#x2032;), where measurements may be expected. The wider the measurement distribution, the larger the measurement noise. During the trial simulation, monitoring measurements are sampled from the measurement distribution. This model can be extended to include noise also in the measurement process of covariates, for example to reflect genotyping errors of the VKORC1 or CYP2C9 genes. This is however not considered in this article.</p>
<p>In the 4<sup>th</sup> panel of <xref ref-type="fig" rid="F3">Figure 3</xref>, we illustrate INR measurements sampled from the measurement distribution during warfarin treatment (blue scatter points). The dashed line shows the treatment response simulation of the mechanistic model without measurement noise.</p>
</sec>
</sec>
<sec id="s2-2">
<title>2.2 Implementation of warfarin trial simulation</title>
<p>We develop the warfarin clinical trial model following the framework for MIPD trial simulation introduced in <xref ref-type="sec" rid="s2-1">Section 2.1</xref>. The mechanistic model is implemented using Wajima et al.&#x2019;s model of the humoral coagulation network (see <xref ref-type="fig" rid="F2">Figure 2</xref>). The model provides a mechanistic description of warfarin&#x2019;s PKPD using a system of nonlinear differential equations<disp-formula id="e6">
<mml:math id="m14">
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>k</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mspace width="1em"/>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi mathvariant="bold">x</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi mathvariant="normal">d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="bold">f</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>a</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(6)</label>
</disp-formula>
<italic>a</italic>
<sub>
<italic>d</italic>
</sub> and <italic>a</italic>
<sub>
<italic>c</italic>
</sub> describe the pharmacokinetics of warfarin and denote the amount of the drug in the dose compartment and the central compartment, respectively. <italic>k</italic>
<sub>
<italic>a</italic>
</sub> denotes the absorption rate and <italic>k</italic>
<sub>
<italic>e</italic>
</sub> denotes the elimination rate. <italic>r</italic>(<italic>t</italic>) denotes the dose rate and implements the dosing regimen. The pharmacodynamics of warfarin are captured by <bold>x</bold>, denoting the 51 remaining states of the model. The prothrombin time is simulated as a function of the states, <inline-formula id="inf9">
<mml:math id="m15">
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi mathvariant="bold">x</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, involving the computation of the fibrin AUC after exposure to 300&#xa0;nM tissue factor. For full details of the mechanistic model, we refer to (<xref ref-type="bibr" rid="B57">Wajima et al., 2009</xref>) and <xref ref-type="sec" rid="s10">Supplementary Appendix S1</xref>. A systems biology markup language (SBML) specification is provided on GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/DavAug/mipd-warfarin">https://github.com/DavAug/mipd-warfarin</ext-link>) for simplified cross-platform implementation of the model.</p>
<p>The population model is implemented using a hybrid of two mixed effects models developed by (<xref ref-type="bibr" rid="B20">Hamberg et al., 2010</xref>; <xref ref-type="bibr" rid="B23">Hartmann et al., 2016</xref>; <xref ref-type="bibr" rid="B22">2020</xref>). <xref ref-type="bibr" rid="B23">Hartmann et al. (2016)</xref> provide a mixed effects model extension of Wajima et al.&#x2019;s model, capturing the treatment response variability emergent from varying production rates of selected coagulation factors, including prothrombin, protein S, protein C and coagulation factors V, VII, IX, X, XI, XII and XIII. In a subsequent publication, they extend their model to include variability explained by covariates, such as the genotypes of the VKORC1 gene and the CYP2C9 gene (<xref ref-type="bibr" rid="B22">Hartmann et al., 2020</xref>). VKORC1 is used to model the variability in warfarin&#x2019;s EC50, while CYP2C9 is used to model the variability in warfarin&#x2019;s elimination rate. In our population model, we modify Hartmann et al.&#x2019;s model further using elements from Hamberg et al.&#x2019;s model to incorporate age and heterozygosity in the genotypes as covariates of the IIV (<xref ref-type="bibr" rid="B20">Hamberg et al., 2010</xref>). This results in a population distribution whose subpopulations, <italic>p</italic>(<italic>&#x3c8;</italic>&#x7c;<italic>&#x3b8;</italic>, <italic>&#x3c7;</italic>), are defined by the age of a patient, one of three VKORC1 genotypes (GG; GA; AA) and one of six CYP2C9 genotypes (&#x2a;1&#x2a;1; &#x2a;1&#x2a;2; &#x2a;1&#x2a;3; &#x2a;2&#x2a;2; &#x2a;2&#x2a;3; &#x2a;3&#x2a;3). For full details of the population model, we refer to <xref ref-type="sec" rid="s10">Supplementary Appendix S1</xref>. The population model parameters, <italic>&#x3b8;</italic>, used to simulate the clinical trial are provided on GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/DavAug/mipd-warfarin">https://github.com/DavAug/mipd-warfarin</ext-link>).</p>
<p>The inter-occasion model is implemented using time-varying vitamin K input rates (see 2<sup>nd</sup> panel in <xref ref-type="fig" rid="F3">Figure 3</xref>). The input rate alterations are assumed to be normally distributed<disp-formula id="e7">
<mml:math id="m16">
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(7)</label>
</disp-formula>with constant mean and standard deviation: <italic>&#x3bc;</italic>
<sub>
<italic>&#x3b7;</italic>
</sub> &#x3d; 1 and <italic>&#x3c3;</italic>
<sub>
<italic>&#x3b7;</italic>
</sub> &#x3d; 0.1. To allow a change of vitamin K consumption over time, we sample a new <italic>&#x3b7;</italic> for each simulation day, such that the altered input rate may be interpreted as the daily average of the vitamin K consumption.</p>
<p>The execution model is implemented using exponentially distributed delays of the administration and monitoring times<disp-formula id="e8">
<mml:math id="m17">
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mrow>
<mml:mi mathvariant="normal">e</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">&#x394;</mml:mi>
<mml:mi>t</mml:mi>
<mml:mo>/</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:math>
<label>(8)</label>
</disp-formula>where <italic>&#x3c4;</italic> &#x3d; 30&#xa0;min denotes the average delay. For each nominal administration time and each nominal monitoring time, we independently sample delays from <italic>p</italic>(&#x394;<italic>t</italic>&#x7c;<italic>&#x3c4;</italic>) and compute the actual times according to Eq. <xref ref-type="disp-formula" rid="e4">4</xref>. If dose administrations and monitoring measurements are scheduled at the same nominal times, we only draw one delay random variable for both events.</p>
<p>The measurement model is implemented using lognormally distributed measurements around the mechanistic model output<disp-formula id="e9">
<mml:math id="m18">
<mml:mi>p</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x3d;</mml:mo>
<mml:mi mathvariant="normal">L</mml:mi>
<mml:mi mathvariant="normal">N</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">&#x7c;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>,</mml:mo>
</mml:math>
<label>(9)</label>
</disp-formula>where <inline-formula id="inf10">
<mml:math id="m19">
<mml:mi>&#x3bc;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mover accent="true">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo>&#x304;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>&#x3c8;</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2032;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the median, and <italic>&#x3c3;</italic> &#x3d; 0.1 denotes the scale of the distribution. This implements a measurement error that scales proportionally with the measured quantity, producing measurement errors of approximately 10% of the mechanistic model output.</p>
</sec>
<sec id="s2-3">
<title>2.3 MIPD methods</title>
<p>We investigate three MIPD models for warfarin treatment individualisation: 1. a neural network regression model (&#x2018;Regression model&#x2019;); 2. a deep reinforcement learning model (&#x2018;Deep RL model&#x2019;); and 3. a pharmacokinetic and pharmacodynamic model (&#x2018;PKPD model&#x2019;). All three models use data specific to the to-be-treated patient to predict individualised dosing regimens. While the details differ substantially between the models, <xref ref-type="fig" rid="F4">Figure 4</xref> illustrates their general workflow. The bottom of the figure shows the to-be-treated patient characterised by covariates, such as clinical and genetic factors. The top of the figure shows the treating physician and the MIPD model. The physician uses the model and the patient characteristics to predict an individualised dose. For some MIPD approaches, this dose can be iteratively refined using measurements of the patient&#x2019;s treatment response. Some models can also predict the full future dosing regimen at each iteration rather than just the next dose. Below, we discuss the three methods investigated in this study in more detail.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Schematic illustration of model-informed precision dosing. The top of the figure shows the treating physician and the MIPD model. The bottom of the figure shows the to-be-treated patient characterised by clinical and genetic factors. The physician uses the MIPD model and the patient characteristics to predict an individualised dose. If the treatment response is monitored over time, the physician can use the treatment response measurements and the MIPD model to iteratively refine the dose predictions.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g004.tif"/>
</fig>
<p>1. Regression model &#x2013; This MIPD approach follows <xref ref-type="bibr" rid="B2">Anderson et al. (2012)</xref> and <xref ref-type="bibr" rid="B53">Verhoef et al. (2013)</xref> and uses a static model of the daily maintenance dose to individualise treatments. The maintenance dose refers to the constant warfarin dose administered daily to maintain a desired INR level. The maintenance dose is modelled as a function of the desired response, <italic>y</italic>&#x2a;, and the patient&#x2019;s covariates, <italic>&#x3c7;</italic>,<disp-formula id="e10">
<mml:math id="m20">
<mml:msup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3c7;</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(10)</label>
</disp-formula>
<italic>d</italic>&#x2a; denotes the maintenance dose. This makes it possible to predict individualised maintenance doses based on the covariates of an individual. In contrast to the generalised workflow in <xref ref-type="fig" rid="F4">Figure 4</xref>, the model does not iteratively update its predictions using monitoring measurements.</p>
<p>The model can be implemented using a variety of regression approaches, including linear regression, spline regression and tree regression (<xref ref-type="bibr" rid="B33">Klein et al., 2009</xref>). In this article, we choose a neural network approach. Neural networks are universal function approximators and can therefore learn to approximate the maintenance dose function in Eq. <xref ref-type="disp-formula" rid="e10">10</xref> from data, even when the relationship between the dose and (<italic>&#x3c7;</italic>, <italic>y</italic>&#x2a;) is nonlinear.</p>
<p>For those who are more familiar with neural network regression, in summary we compose the network of three sequential fully connected layers implemented in PyTorch (<xref ref-type="bibr" rid="B43">Paszke et al., 2019</xref>); the two inner layers of this network are of width 1024 and have ReLU activations. The output layer uses a sigmoid activation. The network is trained on simulated trial data (see <xref ref-type="sec" rid="s3-2">Section 3.2</xref>) to minimise the mean squared error objective function using Adam (<xref ref-type="bibr" rid="B32">Kingma and Ba, 2014</xref>). For full details on the implementation and training of the model, we refer to <xref ref-type="sec" rid="s10">Supplementary Appendix S2</xref>.</p>
<p>2. Deep RL model &#x2013; This MIPD approach follows a deep reinforcement learning approach similar to (<xref ref-type="bibr" rid="B61">Zadeh et al., 2023</xref>) and uses a model of the next-to-administer dose to individualise treatments. The dose is modelled as a function of the covariates and the current monitoring data<disp-formula id="e11">
<mml:math id="m21">
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>d</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3c7;</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(11)</label>
</disp-formula>
<italic>d</italic>
<sub>
<italic>j</italic>
</sub> denotes the dose at time <italic>t</italic>
<sub>
<italic>j</italic>
</sub> and <italic>y</italic>
<sub>
<italic>j</italic>
</sub> denotes the INR measurement at time <italic>t</italic>
<sub>
<italic>j</italic>
</sub>. This makes it possible to iteratively predict individualised dosages based on monitoring data and the covariates of an individual, as illustrated in <xref ref-type="fig" rid="F4">Figure 4</xref>. The target treatment response, <italic>y</italic>&#x2a;, is specified before the training of the model (<xref ref-type="bibr" rid="B61">Zadeh et al., 2023</xref>).</p>
<p>Conceptually, RL learns dosing strategies from trial and error: the model sequentially administers dosages and evaluates the &#x2018;goodness&#x2019; of the dose decisions based on the feedback from the treatment response. The learned dosing strategy can be shown to optimally target the desired treatment response under certain technical assumptions and takes the form of a function for the next-to-administer dose (see Eq. <xref ref-type="disp-formula" rid="e11">11</xref>). We discuss the limitations of these assumptions in <xref ref-type="sec" rid="s3-5">Section 3.5</xref>. While trial and error in a clinical setting raises ethical questions, RL models can also be trained on treatment response emulators (<xref ref-type="bibr" rid="B46">Ribba et al., 2022</xref>). Popular treatment response emulators are PKPD models (<xref ref-type="bibr" rid="B61">Zadeh et al., 2023</xref>). To reduce the number of trial and error iterations needed for convergence of the dosing strategy, RL can be performed in conjunction with function approximators (<xref ref-type="bibr" rid="B7">Baird, 1995</xref>). In this article, we choose a deep neural network as the function approximator &#x2013; an approach commonly referred to as deep RL.</p>
<p>For those who are more familiar with deep RL: we use a DQN model to implement the Deep RL model (<xref ref-type="bibr" rid="B41">Mnih et al., 2013</xref>). We compose the network of four sequential fully connected layers implemented in PyTorch (<xref ref-type="bibr" rid="B43">Paszke et al., 2019</xref>): three hidden layers of widths (256, 128, 64) with ReLU activations, and the output layer of width 58. The outputs of the network are the predicted Q-values (<xref ref-type="bibr" rid="B41">Mnih et al., 2013</xref>). The dose with the maximum Q-value is suggested for administration. Following (<xref ref-type="bibr" rid="B61">Zadeh et al., 2023</xref>), the network is trained online using a PKPD model as a treatment response emulator. Prior to the training, the PKPD model is fitted to simulated trial data (see <xref ref-type="sec" rid="s3-2">Section 3.2</xref>). We train the model to minimise the temporal difference error using Double Q-learning (<xref ref-type="bibr" rid="B52">Van Hasselt et al., 2016</xref>) and the Adam optimiser (<xref ref-type="bibr" rid="B32">Kingma and Ba, 2014</xref>). For full details on the implementation and training of the model, we refer to <xref ref-type="sec" rid="s10">Supplementary Appendix S3</xref>.</p>
<p>3. PKPD model &#x2013; This MIPD approach follows a PKPD modelling approach similar to <xref ref-type="bibr" rid="B19">Hamberg et al. (2015)</xref> and uses a model of the dosing regimen to individualise treatments. The dosing regimen is modelled as a function of the covariates, the desired treatment response and the monitoring data<disp-formula id="e12">
<mml:math id="m22">
<mml:mi>r</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>r</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x3c7;</mml:mi>
<mml:mo>,</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2a;</mml:mo>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>.</mml:mo>
</mml:math>
<label>(12)</label>
</disp-formula>
<inline-formula id="inf11">
<mml:math id="m23">
<mml:msub>
<mml:mrow>
<mml:mi mathvariant="script">D</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">{</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mrow>
<mml:mi>r</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">}</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> denotes the monitoring data available up to and including time <italic>t</italic>
<sub>
<italic>j</italic>
</sub>, where we use <italic>r</italic>
<sub>
<italic>j</italic>
</sub> to denote the administered dosing regimen for <italic>t</italic> &#x3c; <italic>t</italic>
<sub>
<italic>j</italic>
</sub>. The function makes it possible to predict individualised dosing regimens based on monitoring data and the covariates of an individual (see <xref ref-type="fig" rid="F4">Figure 4</xref>). The predictions can be iteratively refined as more monitoring data becomes available.</p>
<p>PKPD modelling is a semi-mechanistic modelling approach which makes use of approximate descriptions of the physiological and pharmacological processes to predict the treatment response dynamics. Conceptually, PKPD modelling is similar to the mechanistic model component of the CT model (see e.g., Wajima et al.&#x2019;s models in <xref ref-type="fig" rid="F2">Figure 2</xref>), but generally model the biological processes in lower detail. An example PKPD model is illustrated in <xref ref-type="sec" rid="s10">Supplementary Figure S4.7</xref> in <xref ref-type="sec" rid="s10">Supplementary Appendix S4</xref>. By comparing the predicted and desired treatment responses, this approach is able to determine the optimal dosing regimens for each individual. Analogously to the CT model in <xref ref-type="sec" rid="s2-1-2">Section 2.1.2</xref>, inter-individual variability is described by differences in the model parameters across individuals. Patient-specific parameters are derived from the patient&#x2019;s covariates and monitoring data.</p>
<p>For those who are more familiar with PKPD modelling, in summary we use Hamberg et al.&#x2019;s model (<xref ref-type="bibr" rid="B20">Hamberg et al., 2010</xref>) implemented in chi (<xref ref-type="bibr" rid="B3">Augustin, 2021</xref>) to model the warfarin treatment response dynamics. We fit the model to simulated trial data (see <xref ref-type="sec" rid="s3-2">Section 3.2</xref>) using hierarchical Bayesian inference and the No-U-Turn sampler (NUTS) (<xref ref-type="bibr" rid="B24">Hoffman and Gelman, 2014</xref>) implemented in pints (<xref ref-type="bibr" rid="B9">Clerx et al., 2019</xref>). Individualised dosing regimens are predicted in two steps: 1. The model is fit to an individual&#x2019;s monitoring data using the population model and the individual&#x2019;s covariates as prior knowledge (<xref ref-type="bibr" rid="B37">Maier et al., 2020</xref>); and 2. The dosing regimen is optimised to minimise the mean squared error between the model predictions and the desired treatment response. We use Bayesian inference and pints&#x2019; implementation of the adaptive covariance matrix Markov chain Monte Carlo (ACMC) sampler to fit the model to an individual&#x2019;s data. For the dosing regimen optimisation, we use pints&#x2019; implementation of the covariate matrix adaption evolution strategy (CMA-ES) optimiser (<xref ref-type="bibr" rid="B21">Hansen et al., 2003</xref>). For full details on the implementation and the dosing regimen prediction, we refer to <xref ref-type="sec" rid="s10">Supplementary Appendix S4</xref>.</p>
</sec>
</sec>
<sec sec-type="results|discussion" id="s3">
<title>3 Results and discussion</title>
<p>We simulate three MIPD trials &#x2013; one trial for each MIPD model &#x2013; and analyse their relative strengths and limitations. To this end, we follow the workflow illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref> and, first, fit the MIPD models to simulated clinical trial data to emulate a typical starting point for MIPD trials. The data imitate typical data collected during each of the three phases of clinical trials, and are simulated using the CT model. After the model fitting, we simulate the MIPD trials.</p>
<sec id="s3-1">
<title>3.1 Simulating trial phases prior to MIPD</title>
<p>At present, most clinical trials are conducted not having MIPD in mind. As a result, the data available for the fitting of MIPD models will often not be tailored to the needs of the method. To reflect this practical limitation in our study, we simulate data from three typical phases of clinical trials for the warfarin use case, not taking the data-requirements for MIPD into account. All MIPD models in this article are fitted using only the data from these trials. The simulated trial data as well as the code to reproduce the trials are hosted on GitHub (<ext-link ext-link-type="uri" xlink:href="https://github.com/DavAug/mipd-warfarin">https://github.com/DavAug/mipd-warfarin</ext-link>).</p>
<p>Trial phase I: Phase I trials are primarily used to establish the safety of drugs and often monitor a drug&#x2019;s absorption, distribution, metabolism and elimination (ADME) in a relatively small cohort of patients. We emulate such a trial by mirroring a clinical trial reported in (<xref ref-type="bibr" rid="B18">Hamberg et al., 2007</xref>). In this trial, the pharmacokinetics of <italic>N</italic> &#x3d; 60 individuals is monitored. Each individual is administered with a single 10&#xa0;mg dose of warfarin. Following the nominal administration time, the warfarin concentration in the blood is measured at 10 h, 35 h and 60&#xa0;h after the administration. The data collected during the trial are the warfarin concentration measurements, the dosing regimens and the covariates for each of the 60 individuals. The simulated warfarin concentrations are illustrated in the left panel of <xref ref-type="fig" rid="F5">Figure 5</xref>. The demographics of the cohort are reported in <xref ref-type="sec" rid="s10">Supplementary Table S1</xref>. Pseudo-code outlining the implementation of the trial is presented in <xref ref-type="sec" rid="s10">Supplementary Algorithm S1</xref>.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Pre-MIPD trial workflow. The figure shows the two steps performed prior to the MIPD trial simulation: 1. Simulation of clinical trial data (top); and 2. Fitting of MIPD models (bottom). The CT model is used to simulate the trial data. The simulated data is illustrated in the middle of the figure: phase I (left); phase II (middle); phase III (right). Measurements of the blood warfarin concentration or the INR are illustrated using scatter points. Measurements connected by solid lines are taken from the same individual at different time points. Nominal administration times are illustrated by blue arrows. Dose administrations with individualised, on-the-fly adjustments of the dose amount are highlighted in black. The therapeutic range is illustrated in the middle and right panel using blue dashed lines. The bottom of the figure shows the fitting workflow of the MIPD models: The PKPD model is fitted to the data from all trials; the Regression model is fitted only to the data from clinical trial phase III; and the Deep RL model is fitted to treatment responses simulated with the PKPD model.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g005.tif"/>
</fig>
<p>The panel shows the simulated warfarin concentration measurements. Measurements are illustrated using scatter points. Measurements taken from the same individual are connected using a solid line. The nominal administration time of the warfarin dose is indicated using a blue arrow.</p>
<p>Trial phase II: Phase II trials are primarily used to establish the efficacy of drugs and monitor a drug&#x2019;s pharmacodynamics. We simulate a phase II trial using a design similar to trials reported in (<xref ref-type="bibr" rid="B20">Hamberg et al., 2010</xref>). We monitor the INR response of 100 simulated individuals for 3&#xa0;weeks. Each individual is treated with daily warfarin doses. During the first 3&#xa0;days, all individuals receive the same treatment: <italic>d</italic>
<sub>1</sub> &#x3d; 10&#xa0;mg; <italic>d</italic>
<sub>2</sub> &#x3d; 7.5 mg; and <italic>d</italic>
<sub>3</sub> &#x3d; 5&#xa0;mg. The 4th dose is adjusted for each individual based on the INR treatment response by a medical professional, here emulated using a simple linear heuristic: <italic>d</italic>
<sub>
<italic>j</italic>
</sub> &#x3d; <italic>d</italic>
<sub>
<italic>j</italic>&#x2212;1</sub>&#xa0;<italic>y</italic>&#x2a;/<italic>y</italic>
<sub>
<italic>j</italic>
</sub>, where <italic>y</italic>
<sub>
<italic>j</italic>
</sub> denotes the INR measurement taken just before the <italic>j</italic>th dose administration. This heuristic computes a personalised warfarin dose targeting the desired treatment response, <italic>y</italic>&#x2a;, assuming a linear relationship between the INR and the dose amount. The dose amounts are adjusted three more times for each individual on day 5, 7 and 13 of the trial using the same heuristic. The INR of the individuals is closely monitored during the induction phase of the trial (day 0, day 1, day 2, day 3) and less frequently measured as the trial progresses (day 5, day 7, day 13 and day 20). To emulate safety constrains of real clinical trials, the trial is discontinued when an individual displays three consecutive INR measurements above 5. The data collected during the trial are the INR measurements, the nominal dosing regimens and the covariates for each of the 100 individuals.</p>
<p>The trial was not terminated early for any of the simulated individuals. The simulated INR measurements are illustrated in the middle panel of <xref ref-type="fig" rid="F5">Figure 5</xref>. The demographics of the cohort are reported in <xref ref-type="sec" rid="s10">Supplementary Table S1</xref>. Pseudo-code outlining the implementation of the trial is presented in <xref ref-type="sec" rid="s10">Supplementary Algorithm S2</xref>. The panel shows the simulated INR measurements. Measurements are illustrated using scatter points. Measurements taken from the same individual are connected using a solid line. The nominal administration times of the warfarin doses are indicated using arrows. Black arrows indicate personalised adjustments of the dose amount. The therapuetic range is indicated using blue dashed lines.</p>
<p>Trial phase III: Phase III trials can vary in scope, but tend to involve larger cohorts and have a stronger focus on treatment endpoints. We emulate a phase III trial by mirroring a clinical trial reported in (<xref ref-type="bibr" rid="B33">Klein et al., 2009</xref>). The focus of the trial is to understand the variability of the maintenance warfarin dose. We simulate the trial analogously to trial phase II, but with a larger cohort, <italic>N</italic> &#x3d; 1000, and for a longer duration (8&#xa0;weeks). Following the initial, identical phase of the trial, the daily doses are adjusted two more times on day 27 and 34. During the final 3&#xa0;weeks of the trial, the dose amounts remain unchanged to guarantee that individuals equilibrate to the maintenance treatment response by the end of the trial. To emulate safety constrains of real clinical trials, the trial is discontinued when an individual displays three consecutive INR measurements above 5. The data collected during the trial are the INR measurements at the end of the trial (day 55), the maintenance warfarin doses and the covariates for each of the 1000 individuals.</p>
<p>The trial was not terminated early for any of the simulated individuals. The simulated maintenance doses and INR measurements are illustrated in the right panel of <xref ref-type="fig" rid="F5">Figure 5</xref>. The demographics of the cohort are reported in <xref ref-type="sec" rid="s10">Supplementary Table S1</xref>. Pseudo-code outlining the implementation of the trial is presented in <xref ref-type="sec" rid="s10">Supplementary Algorithm S3</xref>. The panel shows the maintenance dose on the <italic>x</italic>-axis and the measurement of the INR on the <italic>y</italic>-axis. Measurements are illustrated using scatter points. The therapEUtic range is indicated using blue dashed lines.</p>
</sec>
<sec id="s3-2">
<title>3.2 Fitting the MIPD models</title>
<p>We fit the MIPD models to the simulated clinical trial data, as illustrated in <xref ref-type="fig" rid="F5">Figure 5</xref>. The fitting is the second step of the MIPD trial simulation workflow (see <xref ref-type="fig" rid="F1">Figure 1</xref>). Only one of the models &#x2013; the PKPD model &#x2013; can be fitted to all of the available clinical trial data. The Regression model is fitted using the data from the phase III trial. The Deep RL model is trained indirectly on the trial data through simulations from the fitted PKPD model. For details on the fitting, we refer to <xref ref-type="sec" rid="s2-3">Section 2.3</xref>, <xref ref-type="sec" rid="s10">Supplementary Appendix S2</xref>, <xref ref-type="sec" rid="s10">Supplementary Appendix S3</xref> and <xref ref-type="sec" rid="s10">Supplementary Appendix S4</xref>.</p>
</sec>
<sec id="s3-3">
<title>3.3 Simulating the MIPD trials</title>
<p>We simulate one MIPD trial for each of the three methods in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>. All trials are conducted using the same cohort. The cohort includes <italic>N</italic> &#x3d; 1000 individuals. The demographics of the cohort are modelled after a trial reported in (<xref ref-type="bibr" rid="B20">Hamberg et al., 2010</xref>) and are visualised in <xref ref-type="fig" rid="F6">Figure 6</xref>. The figure shows the CYP2C9 genotype distribution, the VKORC1 genotype distribution and the age distribution in the cohort.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Demographics of MIPD trial cohort. The figure shows the CYP2C9 genotype distribution <bold>(A)</bold>, the VKORC1 genotype distribution <bold>(B)</bold> and the age distribution <bold>(C)</bold> of the MIPD trial cohort. The cohort includes 1000 simulated individuals.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g006.tif"/>
</fig>
<p>In each trial, individuals are treated with daily warfarin doses for 19 days. The dose amounts are individualised using the respective MIPD model, as described in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>, and target a treatment response of <italic>y</italic>&#x2a; &#x3d; 2.5. The data available for the individualisation are the INR measurements taken daily before each dose administration and the covariates of the individual. The Regression model individualises the treatments by predicting the maintenance dose based on the covariates of the patients. This maintenance dose is administered every day throughout the trial. The Deep RL model and the PKPD model predict the warfarin doses iteratively based on a patient&#x2019;s covariates and INR measurements. In contrast to the simulated trials in the previous section, we do not emulate safety constrains for the MIPD trials in order to expose possible weaknesses of the models more clearly. Pseudo-code outlining the implementation of the trial is presented in <xref ref-type="sec" rid="s10">Supplementary Algorithm S4</xref>.</p>
</sec>
<sec id="s3-4">
<title>3.4 Results of the MIPD trials</title>
<p>We use three metrics to quantify the success, the safety, and the efficiency of the dosing regimen individualisation: the maintenance INR; the peak INR; and the time in the therapeutic range (TTR). The success of the individualisation is quantified using the maintenance INR measured on the last day of the trial. INR measurements inside the therapeutic range indicate success, while measurements outside the therapeutic range indicate poor dosing regimen individualisation. The safety of the individualisation is quantified using the largest INR measurement recorded during the trial. This peak of the INR response indicates the risk for major bleeding events while transitioning into maintenance treatment. The efficiency of the individualisation is quantified using the TTR, i.e., the number of INR measurements inside the therapeutic range. For successful individualisations, the TTR indicates how quickly the desired treatment response has been achieved.</p>
<p>The results of the trials are visualised in <xref ref-type="fig" rid="F7">Figure 7</xref>. Row 1 shows the results for the Regression model, row 2 shows the results for the Deep RL model, and row 3 shows the results for the PKPD model. The left panel shows the maintenance INR distribution in the cohort. INRs inside the therapeutic range (see blue dashed lines) are highlighted in blue and INRs outside the therapeutic range are highlighted in red. The target INR is illustrated using a black dashed line. The middle panel shows the peak INR distribution in the cohort. The right panel shows the TTR distribution across individuals. The target TTR is illustrated using a dashed line.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>MIPD trial results. The figure shows the outcome of three MIPD trials conducted with an identical cohort of size <italic>N</italic> &#x3d; 1000 with different MIPD models. The top row shows the results for the Regression model; the middle row shows the results for the Deep RL model; and the bottom row shows the results for the PKPD model. The outcome of the trials is illustrated using three metrics: the maintenance INR measured on the last day of the trial <bold>(A)</bold>; the largest INR value measured during the trial <bold>(B)</bold>; and the time in the therapeutic range <bold>(C)</bold>. The panels show the distributions of these metrics across individuals. The therapeutic range is illustrated using blue dashed lines. INR measurements inside the therapeutic range are highlighted in blue, and INR measurements outside the therapeutic range are highlighted in red. Target values of the dosing regimen individualisation are visualised using black dashed lines.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g007.tif"/>
</fig>
<p>The maintenance INR distributions (left panel) show that in a time span of 19&#xa0;days all three MIPD methods are able to successfully target the therapeutic window for a large number of individuals. The Regression model successfully individualises the dosing regimen for 47.0% of the individuals, while the Deep RL model and the PKPD model have success rates of 65.8% and 75.1%, respectively (see blue histograms). For the remaining individuals, the severity of the failed dosing regimen individualisation varies substantially. For the Regression model, 149 individuals display maintenance INRs above 4 with the largest value being 7.51. For the Deep RL model, only 32 individuals exceed maintenance INRs of 4. However, the largest maintenance INR is 29.03 &#x2014; a value almost 4 times larger than for the Regression model, raising serious safety concerns. For the PKPD model the largest maintenance INR is 3.90. 97.8% of the individuals display maintenance INR measurements less than 3.5, showing that the PKPD model is able to most consistently achieve maintenance INRs close to the therapeutic window.</p>
<p>The peak INR distributions (middle panel) show that, for the majority of the cohort, the methods are not able to individualise the dosing regimens without overshooting the therapeutic window. In the Regression model trial, 67.1% of the individuals display peak INR measurements above the therapeutic range. The Deep RL model controls the treatment response marginally better, overshooting the therapeutic range for 65.1% of the individuals. The PKPD model misses the therapeutic window for almost all individuals (95.9%) before reaching maintenance treatment. However, the panel also shows that the largest INR value measured in the PKPD model trial is substantially smaller than the largest INR values in the other two trials (Regression model: 7.99; Deep RL model: 29.03; PKPD model: 4.91), indicating that the PKPD model is the safest MIPD approach among the tested models.</p>
<p>The TTR distributions (right panel) show that the time spent inside the therapeutic range varies between individuals and MIPD approaches. For the Regression model, the median TTR is 45% of the trial duration, with TTRs ranging between 0% and 85%. The other two methods achieve substantially larger TTRs across individuals. For the Deep RL model, the median TTR is 68% of the trial duration, with individual values ranging between 5% and 95%. The PKPD model achieves a median TTR of 74%, with a minimum TTR of 26% and a maximum TTR of 100%. This shows that across individuals the PKPD model takes the least time to successfully reach the therapeutic window.</p>
</sec>
<sec id="s3-5">
<title>3.5 Degrees of dosing regimen individualisation</title>
<p>The simulated trials in <xref ref-type="sec" rid="s3-4">Section 3.4</xref> show that different MIPD approaches have different strengths and limitations. In this section, we study the dosing strategies of the models in more detail to gain a better understanding about their practical and methodological differences. We pay particular attention to attributing generic strengths and limitations to the methodology and specific strengths and limitations to our implementation.</p>
<p>We investigate the dosing strategies by studying the dose decisions suggested by each of the models for three representative individuals from the simulated trials in <xref ref-type="sec" rid="s3-4">Section 3.4</xref>. The first individual, with ID 716, is characterised by the covariates <italic>&#x3c7;</italic>
<sub>
<italic>A</italic>
</sub> &#x3d; (&#x2a;1&#x2a;1, GG, 50). The other two individuals (ID 269; ID 305) are both characterised by the covariates, <italic>&#x3c7;</italic>
<sub>
<italic>B</italic>
</sub> &#x3d; (&#x2a;1&#x2a;2, GA, 50). The different dosing strategies and treatment responses are visualised in <xref ref-type="fig" rid="F8">Figure 8</xref>. The figure shows the doses administered during the trials in the top panel and the corresponding treatment response measurements in the bottom panel. Doses or measurements belonging to the same individual are connected using solid lines. The therapeutic range is illustrated using dashed lines. The left panel shows the trial results for the Regression model, the middle panel shows the trial results for the Deep RL model, and the right panel shows the trial results for the PKPD model.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Degrees of dosing regimen individualisation. The figure shows the dosing regimen individualisations achieved by the MIPD models in the simulated MIPD trials for three representative individuals. The left panel shows the results for the Regression model, the middle panel shows the results for the Deep RL model, and the right panel shows the results for the PKPD model. The top row illustrates the administered dose amounts and the bottom panel illustrates the INR monitoring data. Dose amounts or measurements belonging to the same individual are connected using solid lines. The therapeutic range is illustrated using dashed lines. One of the individuals (ID 716) is characterised by the covariates <italic>&#x3c7;</italic>
<sub>
<italic>A</italic>
</sub> &#x3d; (&#x2a;1&#x2a; 1, GG, 50). The other two individuals (ID 269; ID 305) are both characterised by the covariates <italic>&#x3c7;</italic>
<sub>
<italic>B</italic>
</sub> &#x3d; (&#x2a;1&#x2a; 2, GA, 50).</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g008.tif"/>
</fig>
<sec id="s3-5-1">
<title>3.5.1 The Regression model</title>
<p>The left panel of the figure shows that the Regression model predicts a maintenance dose of 9&#xa0;mg for the individual with ID 716 (black scatter points), and a maintenance dose of 4.5&#xa0;mg for the other two individuals (see top left panel in <xref ref-type="fig" rid="F8">Figure 8</xref>). These maintenance doses are administered daily throughout the trial. For the individual with ID 305, the maintenance INR measurement at the end of the trial is inside the therapeutic range, while the maintenance INR measurements for the other two individuals overshoot the therapeutic range (see bottom left panel in <xref ref-type="fig" rid="F8">Figure 8</xref>). Notably, the individual with the successful individualisation (ID 305) has the same covariates as one of the individuals with the failed individualisation (ID 269).</p>
<p>This illustrates both a strength and a limitation of the Regression model: a strength of the model is that it is able to predict individualised dosages using information only about the covariates of individuals, making the model easier to implement in clinical practice than monitoring-based approaches. However, the figure also shows that dosing regimens exclusively derived from covariates will, at best, successfully target the desired treatment response for an average individual characterised by the covariates. Inter-individual differences that are not explained by covariates are not accounted for. In this case, the individual with ID 305 happens to respond similarly to an average individual<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref> with the covariates, <italic>&#x3c7;</italic>
<sub>
<italic>B</italic>
</sub>, resulting in a successful dosing regimen individualisation. The individual with ID 269, on the other hand, responds more strongly to warfarin than an average individual<xref ref-type="fn" rid="fn2">
<sup>2</sup>
</xref>, yielding a maintenance INR above the therapeutic range. Not being represented by an average individual also explains the failed individualisation for the individual with ID 716. The inability to account for unexplained IIV is a generic limitation of approaches exclusively based on covariates.</p>
<p>In addition to this limited ability to account for IIV, the figure also shows that the Regression model is incapable of accounting for inter-occasional variation and differences in the execution of the treatment. This limitation is, again, a direct consequence of exclusively using covariates for the dosing regimen individualisation. Without quantifying the individual-specific IOV and EV, predicted dosing regimens can, at best, be successful when the IOV and the EV of the treated individual happen to be close to the average IOV and EV observed in the dataset used for the model fitting.</p>
<p>Another limitation, illustrated in <xref ref-type="fig" rid="F8">Figure 8</xref>, is that the Regression model cannot account for treatment response delays. The model only predicts maintenance dosages, and therefore provides no guidance for individualisation of the induction phase of the treatment. In our implementation, we choose to overcome this limitation of the model by administering the predicted maintenance dosages from the beginning of the trial, not attempting to individualise induction dosing regimens without model guidance. This has the consequence that treatment responses take time to reach their maintenance level, limiting the efficiency attainable by the model. From <xref ref-type="fig" rid="F8">Figure 8</xref>, we can, for example, see that the INR measurements of the individual with ID 305 reach the therapeutic range for the first time on day 8 of the treatment. After that, two more measurements are outside the therapeutic range, giving rise to a TTR of 10 days during the 19&#xa0;days trial. Together with the top right panel in <xref ref-type="fig" rid="F7">Figure 7</xref>, this indicates that even when the Regression model individualises maintenance dosages successfully, it achieves, at best, an average TTR of around 55%&#x2013;65%. This limit to the efficiency is specific to the treatment response delay of warfarin and our implementation of the Regression model.</p>
<p>A summary of the strengths and limitations specific to the Regression model are presented in the left column of <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of the strengths and limitations of the MIPD models. The properties of the models are grouped into data-related properties (rows 1&#x2013;3), variability-related properties (rows 4&#x2013;8), and strategy-related properties (rows 9&#x2013;13). The parentheses around the property of the PKPD model in the top right corner indicate that the model, as defined in <xref ref-type="sec" rid="s2-3">Section 2.3</xref>, can be used without covariates, but, in the simulated MIPD trial in <xref ref-type="sec" rid="s3-4">Section 3.4</xref>, we did not explore this possibility. Similarly, parentheses around the property below indicate that the PKPD model can be used without the use of monitoring data.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Regression model</th>
<th align="left">Deep RL model</th>
<th align="left">PKPD model</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="left">Requires covariates</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2713;</td>
<td align="left">(&#x2713;)</td>
</tr>
<tr>
<td align="left">Requires monitoring</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
<td align="left">(&#x2713;)</td>
</tr>
<tr>
<td align="left">Requires indefinite monitoring</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2717;</td>
</tr>
<tr>
<td align="left">Accounts for explained IIV</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2713;</td>
</tr>
<tr>
<td align="left">Accounts for unexplained IIV</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2713;</td>
</tr>
<tr>
<td align="left">Accounts for IOV</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2713;</td>
</tr>
<tr>
<td align="left">Accounts for EV</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2713;</td>
</tr>
<tr>
<td align="left">Accounts for treatment response delays</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
</tr>
<tr>
<td align="left">Robust to unseen treatment responses</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
</tr>
<tr>
<td align="left">Robust to model misspecification</td>
<td align="left">&#x2713;</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2717;</td>
</tr>
<tr>
<td align="left">Learns individual-specific dosing regimens</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2717;</td>
<td align="left">&#x2713;</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3-5-2">
<title>3.5.2 The Deep RL model</title>
<p>In comparison, the Deep RL model is able to predict more individualised dosing regimens than the Regression model (see middle panel in <xref ref-type="fig" rid="F8">Figure 8</xref>). The panel shows that for all three individuals, the model begins the treatment with increased warfarin dosages of more than 20&#xa0;mg. During the maintenance phase, it reduces the dosages. For the individual with covariates <italic>&#x3c7;</italic>
<sub>
<italic>A</italic>
</sub> (ID 716), the model alternates between administering dosages of 12&#xa0;mg and 3&#xa0;mg. For the other two individuals with covariates <italic>&#x3c7;</italic>
<sub>
<italic>B</italic>
</sub>, the model administers a constant dose of 3&#xa0;mg, from which it only occasionally deviates for one of the individuals (ID 269). Overall, the figure shows that the Deep RL model is more successful in individualising the dosing regimens of the individuals, as indicated by their maintenance INR measurements at the end of the trial. The individualisation of the induction phase reduces the time needed to reach the therapeutic range, leading to all three individuals displaying INR measurements inside the therapeutic range within the first four treatment days.</p>
<p>The main reason for the improved performance of the Deep RL model is the use of feedback control from the monitoring data in addition to the covariate information. The model derives predictions from both monitoring data and covariates using the dose function, defined in Eq. <xref ref-type="disp-formula" rid="e11">11</xref>, which predicts the next-to-administer dose based on the most recently measured INR value and the covariates of the to-be-treated individual. We visualise the predictions of the dose function for the three individuals in <xref ref-type="fig" rid="F9">Figure 9</xref>. For illustrative purposes, the figure focuses on INR values between 0.5 and 7. The predicted dose values are illustrated using a black line for individuals with the covariates <italic>&#x3c7;</italic>
<sub>
<italic>A</italic>
</sub> and a red-blue line for individuals with the covariates <italic>&#x3c7;</italic>
<sub>
<italic>B</italic>
</sub>. The target treatment response is illustrated using a black dashed line, and the therapeutic range is indicated using blue dashed lines.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Dosing strategy of the Deep RL model. The figure illustrates the dose function of the Deep RL model, defined in Eq. <xref ref-type="disp-formula" rid="e11">11</xref>, for a range of possible INR monitoring measurements and two sets of covariates: <italic>&#x3c7;</italic>
<sub>
<italic>A</italic>
</sub> &#x3d; (&#x2a;1&#x2a;1, GG, 50) (black line); and <italic>&#x3c7;</italic>
<sub>
<italic>B</italic>
</sub> &#x3d; (&#x2a;1&#x2a;2, GA, 50) (red-blue line). The function determines the next-to-administer dose based on the most recently measured INR value (bottom axis) and the covariates of an individual. The target treatment response is illustrated using a black dashed line. The therapeutic range is indicated using blue dashed lines.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g009.tif"/>
</fig>
<p>The figure shows that the Deep RL model has a clear strategy for INR monitoring measurements below 4, and a less clear strategy for INR measurements above 4. We will first focus on the dose decisions for INRs below 4: for INR measurements below the therapeutic range, the model administers large warfarin doses; for INRs inside the therapeutic range, the model administers intermediate warfarin doses; and for INRs above the therapeutic range, the model administers low warfarin doses. The exact change points and dose amounts are specific to the covariates. For example, for individuals with the <italic>&#x3c7;</italic>
<sub>
<italic>A</italic>
</sub> covariates, the model starts the treatment with a dose of 21.5&#xa0;mg and switches to an intermediate dose of 12&#xa0;mg for INR measurements between 2.1 and 2.7 (see black line in <xref ref-type="fig" rid="F9">Figure 9</xref>). This dosing strategy is consistent with the dose decisions observed in <xref ref-type="fig" rid="F8">Figure 8</xref>.</p>
<p>The dose function illustrates both a strength and a limitation of the Deep RL model (see <xref ref-type="fig" rid="F9">Figure 9</xref>). A strength of the model is that it bases its dose decisions on a simple and interpretable feedback control mechanism: when INR values are too low, it increases the warfarin dose; and when INR values are too high, it decreases the warfarin dose. These dose decisions are tailored to an individual based on the individual&#x2019;s covariates. This strategy leads to a substantially higher success rate and efficiency of the dosing regimen individualisation relative to the Regression model (see <xref ref-type="fig" rid="F7">Figure 7</xref>), as the feedback control enables the model to maintain treatment responses within a desired range, even when unexplained IIV, IOV and EV are present. This strength is generic to reinforcement learning approaches.</p>
<p>However, while the ability to utilise monitoring data for control is a strength, the model&#x2019;s inability to learn from the measurements is a limitation of the Deep RL model, making the approach indefinitely reliant on monitoring data. The reliance on monitoring data is specific to the Deep RL model and is a consequence of its learning approach: the model establishes the dose function (Eq. <xref ref-type="disp-formula" rid="e11">11</xref>) prior to the treatment of individuals (see <xref ref-type="sec" rid="s3-2">Section 3.2</xref>), never individualising the dose function based on the individual-specific monitoring data. Instead, the model treats individuals with dosing strategies exclusively based on covariates, where the monitoring data is only used as a control mechanism to steer treatment responses back to the desired treatment response, if needed (see e.g., the dosing strategy in <xref ref-type="fig" rid="F8">Figure 8</xref> for ID 269). In principle, reinforcement learning approaches can update the dose function using individual-specific monitoring data (<xref ref-type="bibr" rid="B38">Maier et al., 2021</xref>). However, for the Deep RL model, meaningful updates of the dose function are challenging, as the large number of model parameters of its neural network complicates the balance between fine-tuning and overfitting to the limited number of measurements.</p>
<p>The absence of fully individualised dosing strategies explains the observed tendency of the Deep RL model to overshoot the therapeutic range (see middle panel in <xref ref-type="fig" rid="F7">Figure 7</xref>): to successfully target the therapeutic range across individuals with the same set of covariates, the initial warfarin dose predicted by the dose function needs to be high enough, so that even the weakest responders in a subpopulation reach the therapeutic range. This dose leads to INR responses above the therapeutic range for strong responders in the same subpopulation due to the substantial level of warfarin IIV not explained by covariates. The over-treatment of the strong responders is compensated for by reducing the dose for INRs above the therapeutic range so much that even the strongest responders are guaranteed to regress back to the therapeutic range (see large dose steps in <xref ref-type="fig" rid="F9">Figure 9</xref>). This explains the substantial fraction of individuals with peak INRs above the therapeutic range in <xref ref-type="fig" rid="F7">Figure 7</xref>, and points to a general lack of precision of the Deep RL model. This &#x2018;control over precision&#x2019; strategy is a generic limitation of reinforcement learning approaches that do not individualise their dose functions based on the individual-specific monitoring data.</p>
<p>The dose function in <xref ref-type="fig" rid="F9">Figure 9</xref> also demonstrates that the Deep RL model can fail to learn meaningful dosing strategies for the full range of possible INR measurements. When INR measurements become large (<inline-formula id="inf12">
<mml:math id="m24">
<mml:mo>&#x2265;</mml:mo>
<mml:mn>5.05</mml:mn>
</mml:math>
</inline-formula> for <italic>&#x3c7;</italic>
<sub>
<italic>A</italic>
</sub>; and <inline-formula id="inf13">
<mml:math id="m25">
<mml:mo>&#x2265;</mml:mo>
<mml:mn>4.3</mml:mn>
</mml:math>
</inline-formula> for <italic>&#x3c7;</italic>
<sub>
<italic>B</italic>
</sub>), the model suggests increasing the warfarin dose to 5&#xa0;mg, despite the fact that higher warfarin dosages inevitably lead to even higher INR values. The root for these dose decisions lies in the training of the Deep RL model (see <xref ref-type="sec" rid="s3-2">Section 3.2</xref>): large INR monitoring measurements remain under-explored during the training, leading to poorly tested dose decisions for large INR measurements. Those dose decisions can remain without consequence, when the model is able to control treatment responses well enough to never reach large INR values (see middle panel in <xref ref-type="fig" rid="F8">Figure 8</xref>). But when an individual responds more strongly to warfarin than expected, for example due to unexplained IIV, IOV or EV, poorly tested dose decisions for large INR values can cause a failure of the model&#x2019;s feedback control mechanism. This failure explains the severe over-treatments of a few individuals observed in the simulated MIPD trial in <xref ref-type="fig" rid="F7">Figure 7</xref>.</p>
<p>The lack of exploration during the training, despite the use of a standard training procedure (see <italic>&#x3f5;</italic>-greedy policy in <xref ref-type="sec" rid="s10">Supplementary Appendix S3</xref>), is the consequence of a mismatch between the technical assumptions of the Deep RL model and the reality of warfarin treatment responses: the Deep RL model assumes that there is no delay between dose administrations and the feedback from INR measurements, when, in fact, warfarin treatment responses have delays of up to 10&#xa0;days (see e.g., <xref ref-type="fig" rid="F3">Figure 3</xref>). This reduces the ability of the standard training procedure to contribute to the exploration of large INR measurements. The mismatch between the model assumptions and the treatment response delay is generic to reinforcement learning models and difficult to overcome, as the convergence and optimality of reinforcement learning centrally relies on the assumption that transition dynamics can be modelled by a Markov decision process with i.i.d. actions and i.i.d. states (<xref ref-type="bibr" rid="B50">Sutton and Barto, 2018</xref>). For the Deep RL model, this implies that the model has to assume that INR measurements on 1&#xa0;day depend only on the INR measurement and the administered dose on the previous day, i.e., any dose administrations and INR measurements on earlier days cannot be taken into account. Despite those technical constraints, <xref ref-type="bibr" rid="B61">Zadeh et al. (2023)</xref> show that their deep reinforcement learning model is able to improve its performance, when the i.i.d. assumption of the states is explicitly violated and dose decisions are conditioned on more than one recent INR measurement. As a result, the technical assumptions of reinforcement learning, needed for convergence and optimality, may limit the ability of reinforcement learning models to account for treatment response delays, in theory, but in practice, it may be possible to overcome those limitations (<xref ref-type="bibr" rid="B15">Gaon and Brafman, 2020</xref>).</p>
<p>A summary of the strengths and limitations specific to the Deep RL model are presented in the middle column of <xref ref-type="table" rid="T1">Table 1</xref>.</p>
</sec>
<sec id="s3-5-3">
<title>3.5.3 The PKPD model</title>
<p>The right panel in <xref ref-type="fig" rid="F8">Figure 8</xref> shows that the PKPD model achieves the highest degree of dosing regimen individualisation among the tested models. The model begins the treatment for all three individuals with a high dose of 30&#xa0;mg, followed by a gradual reduction of the dose over the following days. For the individual with ID 269, this dose reduction is more rapid than for the other two individuals. Towards the end of the trial, the dosages converge to constant, individual-specific maintenance dosages. The maintenance INRs are located at the upper threshold of the therapeutic range in close proximity to each other, indicating success of the dosing regimen individualisation. All three individuals display INR measurements inside the therapeutic range within the first three treatment days.</p>
<p>The main reason for the good performance of the PKPD model is its use of both covariate information and monitoring data similar to the Deep RL model. The model uses the covariates to predict initial dosing regimens that target the desired treatment response for average individuals. These initial dosing regimens are further individualised based on the individual-specific monitoring data, achieving an increasing degree of individualisation as more monitoring data becomes available. This increasing degree of individualisation provides a distinct advantage over the other two MIPD models. The PKPD model derives predictions from both monitoring data and covariates using the dosing regimen function, defined in Eq. <xref ref-type="disp-formula" rid="e12">12</xref>, which predicts individualised dosing regimens based on all available INR measurements of the to-be-treated individual, the covariates, and the already-administered dosages. We visualise the values of the dosing regimen function for the three individuals in <xref ref-type="fig" rid="F10">Figure 10</xref>. For illustrative purposes, the figure focuses on 4&#xa0;days of the simulated trial: day 1; day 2; day 7; and day 16. The predicted dose values are illustrated using scatter points in two opacity levels: opaque scatter points indicate already administered dosages; and faded scatter points indicate future dose administrations. The treatment day is illustrated using dashed lines.</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Dosing strategy of the PKPD model. The figure illustrates the dosing regimen function of the PKPD model, defined in Eq. <xref ref-type="disp-formula" rid="e12">12</xref>, for the three individuals from <xref ref-type="fig" rid="F8">Figure 8</xref> on 4&#xa0;days of the trial: on the 1st treatment day (panel 1), on the 2nd treatment day (panel 2); on the 7th treatment day (panel 3) and on the 16th treatment day (panel 4). Dosages are illustrated using scatter points in two opacity levels: opaque scatter points for already administered dosages; faded scatter points for future dose administrations. Dosages belonging to the same individual are connected using solid lines. The day of the treatment is illustrated using black dashed lines.</p>
</caption>
<graphic xlink:href="fphar-14-1270443-g010.tif"/>
</fig>
<p>The figure shows that the dosing regimen predictions are iteratively updated as the treatment progresses. On day 1, the model provides rough estimates of the dosing regimens, scheduling maintenance dosages of 8&#xa0;mg (ID 716), 4&#xa0;mg (ID 269), and 5.5&#xa0;mg (ID 305) after an initial induction phase of the treatment. With the next INR measurement on day 2, both the induction dosing regimen as well as the maintenance dosages are updated (see second panel in <xref ref-type="fig" rid="F10">Figure 10</xref>). On day 16, the model predicts fully individualised dosing regimens which are almost identical to the dosing regimens administered in the simulated trial (see right panel in <xref ref-type="fig" rid="F8">Figure 8</xref>).</p>
<p>The dosing regimen function illustrates both a strength and a limitation of the PKPD model (see <xref ref-type="fig" rid="F10">Figure 10</xref>). A strength of the model is that it predicts full dosing regimens from any number of monitoring measurements, making the dosing schedule transparent, foreseeable, and less reliant on frequent or regular monitoring. The prediction of full dosing regimens is enabled by the model&#x2019;s explicit description of the pharmacological processes in terms of a semi-mechanistic model. This permits the prediction of treatment responses, and thus optimal dosing regimens, for any future time points. It also helps the model to account for treatment response delays and nonlinearities of the dose-response relationship (see <xref ref-type="sec" rid="s2-3">Section 2.3</xref>). However, the explicit model of the treatment response bears a risk for model misspecification (<xref ref-type="bibr" rid="B40">Merl&#xe9; et al., 2004</xref>). In particular, neglecting or oversimplifying important treatment response mechanisms can lead to inaccurate treatment response predictions, which, in turn, can impact the quality of the dosing regimen individualisation. For example, the results from the simulated MIPD trial show that the PKPD model tends to administer too large warfarin dosages during the trial, resulting in a systematic bias towards INR measurements larger than the target INR (see bottom panel of <xref ref-type="fig" rid="F7">Figure 7</xref>). This indicates that the PKPD model oversimplifies crucial elements of the treatment response mechanisms, resulting in a tendency to underestimate the treatment response of individuals. The risk for model misspecification is a generic limitation of PKPD modelling, which needs to be mitigated prior to clinical applications, for example by quantifying the structural uncertainty of PKPD models using model selection criteria or probabilistic model averaging (<xref ref-type="bibr" rid="B51">Uster et al., 2021</xref>; <xref ref-type="bibr" rid="B5">Augustin et al., 2022</xref>).</p>
<p>The dosing regimen function in <xref ref-type="fig" rid="F10">Figure 10</xref> also illustrates that the updates of the dosing regimens themselves can result in a bias of the treatment strategy. The comparison between the predicted dosing regimens shows that although more individual-specific monitoring measurements are available on day 7 of the treatment, the maintenance dosages predicted on day 2 are closer to the actual maintenance dosages administered towards the end of the trial. This suggests that the degree of the dosing regimen individualisation can temporarily decrease with the number of monitoring measurements &#x2013; a limitation specific to our implementation of the dosing regimen individualisation. The potential for worse dosing regimens despite more monitoring data is related to the estimation of the individual &#x2013; specific model parameters from the monitoring data: we estimate the model parameters using Bayesian inference (see <xref ref-type="sec" rid="s2-3">Section 2.3</xref>). The result of this estimation is a distribution of parameter values consistent with the data, also known as posterior distribution. In our implementation, we estimate the individual-specific model parameters by the modes of the distribution, also known as maximum <italic>a posteriori</italic> (MAP) estimates. The MAP estimates are a popular choice to reduce posterior distributions to just one set of model parameters (<xref ref-type="bibr" rid="B48">Sheiner et al., 1979</xref>). However, by disregarding the other model parameters that are also consistent with the data, it is possible to introduce biases in the treatment response predictions with consequences for the dosing regimen individualisation. In particular for nonlinear treatment responses, the treatment response predicted with the MAP estimates is, generally, not the treatment response that maximises the predictive probability (<xref ref-type="bibr" rid="B37">Maier et al., 2020</xref>). This increases the risk for inaccurate treatment response predictions.</p>
<p>A bias of the MAP-based treatment response predictions explains the lack of safety observed during the simulated MIPD trial (see bottom panel in <xref ref-type="fig" rid="F7">Figure 7</xref>). The MAP-based predictions tend to underestimate the warfarin treatment response for small INR values which leads to individualised dosing regimens with elevated dose amounts early in the trial. As the treatment progresses, the uncertainty about the model parameters becomes smaller, reducing the error of the MAP estimation, and with it, the bias of the treatment response predictions (see maintenance INR distribution in <xref ref-type="fig" rid="F7">Figure 7</xref>). This bias is also supported by the goodness-of-fit plot from the model fit to the pre-MIPD trial data in <xref ref-type="sec" rid="s3-2">Section 3.2</xref>, where also the predictions with the maximum probability parameters of the population distribution show a tendency to provide biased treatment response predictions (see middle and bottom right panel in <xref ref-type="sec" rid="s10">Supplementary Figure S4.9</xref>). This limitation of our PKPD model can be mitigated by predicting the treatment response with each parameter set that is consistent with the data, i.e., the parameters in the posterior distribution, producing a distribution of treatment responses which reflects the uncertainty in the treatment response predictions (<xref ref-type="bibr" rid="B37">Maier et al., 2020</xref>). The distribution of treatment responses can be optimised to obtain a dosing regimen with less risk for bias (<xref ref-type="bibr" rid="B38">Maier et al., 2021</xref>).</p>
<p>A summary of the strengths and limitations specific to the PKPD model are presented in the right column of <xref ref-type="table" rid="T1">Table 1</xref>.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusion" id="s4">
<title>4 Conclusion</title>
<p>Simulated clinical trials provide a resource-efficient way to test and develop fit-for-purpose models for precision dosing. We show that we can emulate clinical trials using a clinical trial model with five independent model components: 1. a mechanistic model; 2. a population model; 3. an inter-occasion model; 4. an execution model; and 5. a measurement model. Each model component captures a different complexity of clinical practice that challenges the successful individualisation of treatments, ranging from PKPD-related challenges, such as nonlinear and delayed treatment responses, to practical challenges, such as unintentional deviations from nominal dosing schedules (see <xref ref-type="sec" rid="s2-1">Section 2.1</xref>). The modularity of the model components simplifies the development process of the clinical trial model, allowing for independent updates of its components throughout the drug development pipeline. This makes it possible to iteratively improve the trial simulations and develop a promising companion MIPD tool as more understanding and information about the drug under trial becomes available (<xref ref-type="bibr" rid="B44">Polasek et al., 2019</xref>).</p>
<p>Simulating trials for precision dosing of warfarin, we find that different MIPD models have different strengths and limitations. These strengths and limitations can be generic to the methodology or specific to the model implementation. Modelling approaches that predict dosing regimens exclusively based on covariates of the treatment response variability are generically limited when unexplained IIV, IOV and EV are present (see <xref ref-type="sec" rid="s3-5-1">Section 3.5.1</xref>). However, when the majority of the treatment response variability can be explained by covariates, and those covariates are available in clinical practice, such approaches provide an excellent solution to the individualisation of dosing regimens. Otherwise, MIPD models based on monitoring data are better at accounting for treatment response variability, and achieve a higher degree of dosing regimen individualisation in the simulated warfarin trials (see <xref ref-type="sec" rid="s3-4">Sections 3.4</xref>, <xref ref-type="sec" rid="s3-5">3.5</xref>).</p>
<p>But there are also challenges with monitoring-based MIPD approaches. We find that the Deep RL model adopts a &#x2018;control over precision&#x2019; treatment strategy, where doses are adjusted based on the feedback response from the monitoring data with limited foresight (see <xref ref-type="sec" rid="s3-5-2">Section 3.5.2</xref>). This lack of precision makes the model reliant on indefinite monitoring and limits its ability to account for treatment response delays. Deep reinforcement learning may, nevertheless, provide a good solution to the individualisation of dosing regimens for applications where monitoring is not a challenge, such as for treatments in the intensive care unit (<xref ref-type="bibr" rid="B42">Moore et al., 2004</xref>) or for dosing devices that are physically attached to patients, like insulin pumps (<xref ref-type="bibr" rid="B62">Zhu et al., 2020</xref>).</p>
<p>The PKPD model achieves the highest degree of dosing regimen individualisation among the tested models. The model predicts fully individualised dosing regimens whose level of individualisation increases with the amount of available monitoring data. This indicates that across applications PKPD models are the most promising approach for precision dosing. However, PKPD models are more susceptible to model misspecifications than the other two approaches (see <xref ref-type="sec" rid="s3-5-3">Section 3.5.3</xref>), necessitating a careful evaluation of the predictive uncertainty of the model prior to MIPD, for example by means of model selection criteria or probabilistic model averaging (<xref ref-type="bibr" rid="B5">Augustin et al., 2022</xref>).</p>
<p>Overall we find that MIPD approaches vary in their ability to account for the different sources of treatment response variability, meaning an ideal approach depends on the context. Distinguishing four sources of treatment response variability &#x2013; 1. unexplained IIV; 2. explained IIV; 3. IOV; and 4. EV &#x2013; our results suggest that PKPD modelling is more successful than the other two approaches in cases where the treatment response variability is dominated by unexplained IIV. PKPD models account for unexplained IIV by individualising treatment response predictions from just a small number of monitoring measurements. We expect PKPD models to also be the favourable choice when the treatment response variability is dominated by EV, as the other tested models cannot adapt their predictions to irregular administration and monitoring schedules. However, when IOV dominates the treatment response variability, deep reinforcement learning may perform best, as its &#x2018;control over precision&#x2019; dosing strategy is agnostic to random fluctuations in the treatment response. Lastly, when the treatment response variability is dominated by explained IIV, all three models likely provide similar maintenance predictions, making regression the preferred choice as long as treatment response delays are not a concern.</p>
<p>While the sources of the variability can indicate which MIPD approach may perform best, the volume and type of the available data determine which MIPD approach is possible. PKPD modelling works with both large and limited amounts of data, provided the treatment response mechanisms are well known and the available data can be linked to the processes involved. In contrast, regression and deep reinforcement learning need more data to be feasible but can process more complex data types and do not rely on a mechanistic understanding of the treatment response.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the article/<xref ref-type="sec" rid="s10">Supplementary Material.</xref>
</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>DA: Conceptualization, Data curation, Formal Analysis, Investigation, Methodology, Project administration, Resources, Software, Validation, Visualization, Writing&#x2013;original draft, Writing&#x2013;review and editing. BL: Supervision, Writing&#x2013;review and editing. MR: Supervision, Writing&#x2013;review and editing. KW: Supervision, Writing&#x2013;review and editing. DG: Funding acquisition, Supervision, Writing&#x2013;review and editing.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This work was supported by the UK Engineering and Physical Sciences Research Council (grant number EP/S024093/1); and the Biotechnology and Biological Sciences Research Council (grant number BB/P010008/1). DA acknowledges EPSRC for studentship support via the Doctoral Training Centre in Sustainable Approaches to Biomedical Science: Responsible and Reproducible Research, as well as the Clarendon Fund for studentship support. BL, MR, and DG acknowledge support from the EPSRC Centres for Doctoral Training Programme. DG acknowledge support from a Biotechnology and Biological Sciences Research Council project grant. KW are employees of F. Hoffmann La Roche Ltd. The funders had no role in study design, data collection and analysis, decision to publish, or preparation of the manuscript.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of interest</title>
<p>Author KW was employed by F. Hoffmann-La Roche AG.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fphar.2023.1270443/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fphar.2023.1270443/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>In this case, the close-to-average response can be explained by the individual&#x2019;s model parameters, <italic>&#x3c8;</italic>, being close to the means of the population distribution (Eq. <xref ref-type="disp-formula" rid="e2">2</xref>).</p>
</fn>
<fn id="fn2">
<label>2</label>
<p>Similarly to the above footnote, the stronger-than-average response can be explained by key model parameters being further away from the means of the population distribution.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Abrantes</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>J&#xf6;nsson</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Karlsson</surname>
<given-names>M. O.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>E. I.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Handling interoccasion variability in model-based dose individualization using therapeutic drug monitoring data</article-title>. <source>Br. J. Clin. Pharmacol.</source> <volume>85</volume>, <fpage>1326</fpage>&#x2013;<lpage>1336</lpage>. <pub-id pub-id-type="doi">10.1111/bcp.13901</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Anderson</surname>
<given-names>J. L.</given-names>
</name>
<name>
<surname>Horne</surname>
<given-names>B. D.</given-names>
</name>
<name>
<surname>Stevens</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Woller</surname>
<given-names>S. C.</given-names>
</name>
<name>
<surname>Samuelson</surname>
<given-names>K. M.</given-names>
</name>
<name>
<surname>Mansfield</surname>
<given-names>J. W.</given-names>
</name>
<etal/>
</person-group> (<year>2012</year>). <article-title>A randomized and clinical effectiveness trial comparing two pharmacogenetic algorithms and standard care for individualizing warfarin dosing (coumagen-ii)</article-title>. <source>Circulation</source> <volume>125</volume>, <fpage>1997</fpage>&#x2013;<lpage>2005</lpage>. <pub-id pub-id-type="doi">10.1161/CIRCULATIONAHA.111.070920</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Augustin</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <source>Chi - an open source python package for treatment response modelling</source>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Augustin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Lambert</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Walz</surname>
<given-names>A.-C.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gavaghan</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Filter inference: a scalable nonlinear mixed effects inference approach for snapshot time series data</article-title>. <source>PLOS Comput. Biol.</source> <volume>19</volume>, <fpage>e1011135</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1011135</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Augustin</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Walz</surname>
<given-names>A.-C.</given-names>
</name>
<name>
<surname>Wang</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lambert</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Clerx</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>M.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Treatment response prediction: is model selection unreliable?</article-title> <source>bioRxiv</source>, <fpage>2022</fpage>&#x2013;<lpage>2103</lpage>. <pub-id pub-id-type="doi">10.1101/2022.03.19.483454</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azer</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kaddi</surname>
<given-names>C. D.</given-names>
</name>
<name>
<surname>Barrett</surname>
<given-names>J. S.</given-names>
</name>
<name>
<surname>Bai</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>McQuade</surname>
<given-names>S. T.</given-names>
</name>
<name>
<surname>Merrill</surname>
<given-names>N. J.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>History and future perspectives on the discipline of quantitative systems pharmacology modeling and its applications</article-title>. <source>Front. physiology</source> <volume>12</volume>, <fpage>637999</fpage>. <pub-id pub-id-type="doi">10.3389/fphys.2021.637999</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Baird</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>1995</year>). &#x201c;<article-title>Residual algorithms: reinforcement learning with function approximation</article-title>,&#x201d; in <source>Machine learning proceedings 1995</source> (<publisher-name>Elsevier</publisher-name>), <fpage>30</fpage>&#x2013;<lpage>37</lpage>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Broeker</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Nardecchia</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Klinker</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Derendorf</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Day</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Marriott</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Towards precision dosing of vancomycin: a systematic evaluation of pharmacometric models for bayesian forecasting</article-title>. <source>Clin. Microbiol. Infect.</source> <volume>25</volume>, <fpage>1286</fpage>. <pub-id pub-id-type="doi">10.1016/j.cmi.2019.02.029</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clerx</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Robinson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lambert</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lei</surname>
<given-names>C. L.</given-names>
</name>
<name>
<surname>Ghosh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Mirams</surname>
<given-names>G. R.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Probabilistic inference on noisy time series (PINTS)</article-title>. <source>J. Open Res. Softw.</source> <volume>7</volume>, <fpage>23</fpage>. <pub-id pub-id-type="doi">10.5334/jors.252</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Darwich</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Ogungbenro</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Vinks</surname>
<given-names>A. A.</given-names>
</name>
<name>
<surname>Powell</surname>
<given-names>J. R.</given-names>
</name>
<name>
<surname>Reny</surname>
<given-names>J.-L.</given-names>
</name>
<name>
<surname>Marsousi</surname>
<given-names>N.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Why has model-informed precision dosing not yet become common clinical reality? lessons from the past and a roadmap for the future</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>101</volume>, <fpage>646</fpage>&#x2013;<lpage>656</lpage>. <pub-id pub-id-type="doi">10.1002/cpt.659</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Darwich</surname>
<given-names>A. S.</given-names>
</name>
<name>
<surname>Polasek</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Aronson</surname>
<given-names>J. K.</given-names>
</name>
<name>
<surname>Ogungbenro</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Wright</surname>
<given-names>D. F.</given-names>
</name>
<name>
<surname>Achour</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Model-informed precision dosing: background, requirements, validation, implementation, and forward trajectory of individualizing drug therapy</article-title>. <source>Annu. Rev. Pharmacol. Toxicol.</source> <volume>61</volume>, <fpage>225</fpage>&#x2013;<lpage>245</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-pharmtox-033020-113257</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dirks</surname>
<given-names>N. L.</given-names>
</name>
<name>
<surname>Meibohm</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Population pharmacokinetics of therapeutic monoclonal antibodies</article-title>. <source>Clin. Pharmacokinet.</source> <volume>49</volume>, <fpage>633</fpage>&#x2013;<lpage>659</lpage>. <pub-id pub-id-type="doi">10.2165/11535960-000000000-00000</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<collab>FDA</collab> (<year>2010</year>). <source>Coumadin&#xae; tablets (warfarin sodium tablets, usp) crystalline coumadin&#xae; for injection (warfarin sodium for injection, usp)</source>
<comment>. <ext-link ext-link-type="uri" xlink:href="https://www.accessdata.fda.gov/drugsatfda_docs/label/2010/009218s108lbl.pdf">https://www.accessdata.fda.gov/drugsatfda_docs/label/2010/009218s108lbl.pdf</ext-link>. Accessed: 2022-October-03</comment>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gage</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Eby</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Johnson</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Deych</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Rieder</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Ridker</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2008</year>). <article-title>Use of pharmacogenetic and clinical factors to predict the therapeutic dose of warfarin</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>84</volume>, <fpage>326</fpage>&#x2013;<lpage>331</lpage>. <pub-id pub-id-type="doi">10.1038/clpt.2008.10</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gaon</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Brafman</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Reinforcement learning with non-markovian rewards</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>34</volume>, <fpage>3980</fpage>&#x2013;<lpage>3987</lpage>. <pub-id pub-id-type="doi">10.1609/aaai.v34i04.5814</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gill</surname>
<given-names>K. L.</given-names>
</name>
<name>
<surname>Machavaram</surname>
<given-names>K. K.</given-names>
</name>
<name>
<surname>Rose</surname>
<given-names>R. H.</given-names>
</name>
<name>
<surname>Chetty</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Potential sources of inter-subject variability in monoclonal antibody pharmacokinetics</article-title>. <source>Clin. Pharmacokinet.</source> <volume>55</volume>, <fpage>789</fpage>&#x2013;<lpage>805</lpage>. <pub-id pub-id-type="doi">10.1007/s40262-015-0361-4</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gong</surname>
<given-names>I. Y.</given-names>
</name>
<name>
<surname>Tirona</surname>
<given-names>R. G.</given-names>
</name>
<name>
<surname>Schwarz</surname>
<given-names>U. I.</given-names>
</name>
<name>
<surname>Crown</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Dresser</surname>
<given-names>G. K.</given-names>
</name>
<name>
<surname>LaRue</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Prospective evaluation of a pharmacogenetics-guided warfarin loading and maintenance dose regimen for initiation of therapy</article-title>. <source>Blood, J. Am. Soc. Hematol.</source> <volume>118</volume>, <fpage>3163</fpage>&#x2013;<lpage>3171</lpage>. <pub-id pub-id-type="doi">10.1182/blood-2011-03-345173</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamberg</surname>
<given-names>A.-K.</given-names>
</name>
<name>
<surname>Dahl</surname>
<given-names>M.-L.</given-names>
</name>
<name>
<surname>Barban</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Scordo</surname>
<given-names>M. G.</given-names>
</name>
<name>
<surname>Wadelius</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pengo</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2007</year>). <article-title>A pk&#x2013;pd model for predicting the impact of age, cyp2c9, and vkorc1 genotype on individualization of warfarin therapy</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>81</volume>, <fpage>529</fpage>&#x2013;<lpage>538</lpage>. <pub-id pub-id-type="doi">10.1038/sj.clpt.6100084</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamberg</surname>
<given-names>A.-K.</given-names>
</name>
<name>
<surname>Hellman</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dahlberg</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Jonsson</surname>
<given-names>E. N.</given-names>
</name>
<name>
<surname>Wadelius</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>A bayesian decision support tool for efficient dose individualization of warfarin in adults and children</article-title>. <source>BMC Med. Inf. Decis. Mak.</source> <volume>15</volume>, <fpage>7</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1186/s12911-014-0128-0</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hamberg</surname>
<given-names>A.-K.</given-names>
</name>
<name>
<surname>Wadelius</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Lindh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Dahl</surname>
<given-names>M.-L.</given-names>
</name>
<name>
<surname>Padrini</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Deloukas</surname>
<given-names>P.</given-names>
</name>
<etal/>
</person-group> (<year>2010</year>). <article-title>A pharmacometric model describing the relationship between warfarin dose and inr response with respect to variations in cyp2c9, vkorc1, and age</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>87</volume>, <fpage>727</fpage>&#x2013;<lpage>734</lpage>. <pub-id pub-id-type="doi">10.1038/clpt.2010.37</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hansen</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>S. D.</given-names>
</name>
<name>
<surname>Koumoutsakos</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2003</year>). <article-title>Reducing the time complexity of the derandomized evolution strategy with covariance matrix adaptation (cma-es)</article-title>. <source>Evol. Comput.</source> <volume>11</volume>, <fpage>1</fpage>&#x2013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1162/106365603321828970</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hartmann</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Biliouris</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lesko</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Nowak-G&#xf6;ttl</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Trame</surname>
<given-names>M. N.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Quantitative systems pharmacology model-based predictions of clinical endpoints to optimize warfarin and rivaroxaban anti-thrombosis therapy</article-title>. <source>Front. Pharmacol.</source> <volume>11</volume>, <fpage>1041</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2020.01041</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hartmann</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Biliouris</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Lesko</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Nowak-G&#xf6;ttl</surname>
<given-names>U.</given-names>
</name>
<name>
<surname>Trame</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Quantitative systems pharmacology model to predict the effects of commonly used anticoagulants on the human coagulation network</article-title>. <source>CPT pharmacometrics Syst. Pharmacol.</source> <volume>5</volume>, <fpage>554</fpage>&#x2013;<lpage>564</lpage>. <pub-id pub-id-type="doi">10.1002/psp4.12111</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hoffman</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Gelman</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2014</year>). <article-title>The no-u-turn sampler: adaptively setting path lengths in Hamiltonian Monte Carlo</article-title>. <source>J. Mach. Learn. Res.</source> <volume>15</volume>, <fpage>1593</fpage>&#x2013;<lpage>1623</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holford</surname>
<given-names>N. H.</given-names>
</name>
<name>
<surname>Kimko</surname>
<given-names>H. C.</given-names>
</name>
<name>
<surname>Monteleone</surname>
<given-names>J. P.</given-names>
</name>
<name>
<surname>Peck</surname>
<given-names>C. C.</given-names>
</name>
</person-group> (<year>2000</year>). <article-title>Simulation of clinical trials</article-title>. <source>Annu. Rev. Pharmacol. Toxicol.</source> <volume>40</volume>, <fpage>209</fpage>&#x2013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.1146/annurev.pharmtox.40.1.209</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Holford</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ma</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Ploeger</surname>
<given-names>B.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Clinical trial simulation: a review</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>88</volume>, <fpage>166</fpage>&#x2013;<lpage>182</lpage>. <pub-id pub-id-type="doi">10.1038/clpt.2010.114</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Caudle</surname>
<given-names>K. E.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Whirl-Carrillo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Stein</surname>
<given-names>C. M.</given-names>
</name>
<name>
<surname>Scott</surname>
<given-names>S. A.</given-names>
</name>
<etal/>
</person-group> (<year>2017</year>). <article-title>Clinical pharmacogenetics implementation consortium (cpic) guideline for pharmacogenetics-guided warfarin dosing: 2017 update</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>102</volume>, <fpage>397</fpage>&#x2013;<lpage>404</lpage>. <pub-id pub-id-type="doi">10.1002/cpt.668</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Johnson</surname>
<given-names>J. A.</given-names>
</name>
<name>
<surname>Gong</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Whirl-Carrillo</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Gage</surname>
<given-names>B. F.</given-names>
</name>
<name>
<surname>Scott</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Stein</surname>
<given-names>C.</given-names>
</name>
<etal/>
</person-group> (<year>2011</year>). <article-title>Clinical pharmacogenetics implementation consortium guidelines for cyp2c9 and vkorc1 genotypes and warfarin dosing</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>90</volume>, <fpage>625</fpage>&#x2013;<lpage>629</lpage>. <pub-id pub-id-type="doi">10.1038/clpt.2011.185</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Karlsson</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Sheiner</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>1993</year>). <article-title>The importance of modeling interoccasion variability in population pharmacokinetic analyses</article-title>. <source>J. Pharmacokinet. Biopharm.</source> <volume>21</volume>, <fpage>735</fpage>&#x2013;<lpage>750</lpage>. <pub-id pub-id-type="doi">10.1007/BF01113502</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Keizer</surname>
<given-names>R. J.</given-names>
</name>
<name>
<surname>Ter Heine</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Frymoyer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Lesko</surname>
<given-names>L. J.</given-names>
</name>
<name>
<surname>Mangat</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Goswami</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Model-informed precision dosing at the bedside: scientific challenges and opportunities</article-title>. <source>CPT pharmacometrics Syst. Pharmacol.</source> <volume>7</volume>, <fpage>785</fpage>&#x2013;<lpage>787</lpage>. <pub-id pub-id-type="doi">10.1002/psp4.12353</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Keutzer</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Simonsson</surname>
<given-names>U. S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Individualized dosing with high inter-occasion variability is correctly handled with model-informed precision dosing&#x2014;using rifampicin as an example</article-title>. <source>Front. Pharmacol.</source> <volume>11</volume>, <fpage>794</fpage>. <pub-id pub-id-type="doi">10.3389/fphar.2020.00794</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kingma</surname>
<given-names>D. P.</given-names>
</name>
<name>
<surname>Ba</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Adam: a method for stochastic optimization</article-title>. <comment>
<italic>arXiv preprint arXiv:1412.6980</italic>
</comment>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Klein</surname>
<given-names>T. E.</given-names>
</name>
<name>
<surname>Altman</surname>
<given-names>R. B.</given-names>
</name>
<name>
<surname>Eriksson</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Gage</surname>
<given-names>B. F.</given-names>
</name>
<name>
<surname>Kimmel</surname>
<given-names>S. E.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>Estimation of the warfarin dose with clinical and pharmacogenetic data</article-title>. <source>N. Engl. J. Med.</source> <volume>360</volume>, <fpage>753</fpage>&#x2013;<lpage>764</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa0809329</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lavielle</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <source>Mixed effects models for the population approach: models, tasks, methods and tools</source>. <edition>1 edn</edition>. <publisher-name>Chapman and Hall/CRC</publisher-name>).</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>X.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Lin</surname>
<given-names>Y.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>Therapeutic drug monitoring of docetaxel by pharmacokinetics and pharmacogenetics: a randomized clinical trial of auc-guided dosing in nonsmall cell lung cancer</article-title>. <source>Clin. Transl. Med.</source> <volume>11</volume>, <fpage>e354</fpage>. <pub-id pub-id-type="doi">10.1002/ctm2.354</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mager</surname>
<given-names>D. E.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Target-mediated drug disposition and dynamics</article-title>. <source>Biochem. Pharmacol.</source> <volume>72</volume>, <fpage>1</fpage>&#x2013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1016/j.bcp.2005.12.041</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maier</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hartung</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>de Wiljes</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Kloft</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Huisinga</surname>
<given-names>W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Bayesian data assimilation to support informed decision making in individualized chemotherapy</article-title>. <source>CPT pharmacometrics Syst. Pharmacol.</source> <volume>9</volume>, <fpage>153</fpage>&#x2013;<lpage>164</lpage>. <pub-id pub-id-type="doi">10.1002/psp4.12492</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Maier</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Hartung</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Kloft</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Huisinga</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>de Wiljes</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Reinforcement learning and bayesian data assimilation for model-informed precision dosing in oncology</article-title>. <source>CPT pharmacometrics Syst. Pharmacol.</source> <volume>10</volume>, <fpage>241</fpage>&#x2013;<lpage>254</lpage>. <pub-id pub-id-type="doi">10.1002/psp4.12588</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Matsumoto</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Oda</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Shoji</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Hanai</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Takahashi</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Fujii</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Clinical practice guidelines for therapeutic drug monitoring of vancomycin in the framework of model-informed precision dosing: a consensus review by the Japanese society of chemotherapy and the Japanese society of therapeutic drug monitoring</article-title>. <source>Pharmaceutics</source> <volume>14</volume>, <fpage>489</fpage>. <pub-id pub-id-type="doi">10.3390/pharmaceutics14030489</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Merl&#xe9;</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Aouimer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Tod</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Impact of model misspecification at design (and/or) estimation step in population pharmacokinetic studies</article-title>. <source>J. Biopharm. Statistics</source> <volume>14</volume>, <fpage>213</fpage>&#x2013;<lpage>227</lpage>. <pub-id pub-id-type="doi">10.1081/BIP-120028516</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Mnih</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kavukcuoglu</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Graves</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Antonoglou</surname>
<given-names>I.</given-names>
</name>
<name>
<surname>Wierstra</surname>
<given-names>D.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <source>Playing atari with deep reinforcement learning</source>. <comment>
<italic>arXiv preprint arXiv:1312.5602</italic>
</comment>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Moore</surname>
<given-names>B. L.</given-names>
</name>
<name>
<surname>Sinzinger</surname>
<given-names>E. D.</given-names>
</name>
<name>
<surname>Quasny</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Pyeatt</surname>
<given-names>L. D.</given-names>
</name>
</person-group> (<year>2004</year>). <article-title>Intelligent control of closed-loop sedation in simulated icu patients</article-title>. <source>Flairs Conf.</source> <volume>109&#x2013;114</volume>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paszke</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Gross</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Massa</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Lerer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Bradbury</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Chanan</surname>
<given-names>G.</given-names>
</name>
<etal/>
</person-group> (<year>2019</year>). <article-title>Pytorch: an imperative style, high-performance deep learning library</article-title>. <source>Adv. neural Inf. Process. Syst.</source> <volume>32</volume>.</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Polasek</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Rayner</surname>
<given-names>C. R.</given-names>
</name>
<name>
<surname>Peck</surname>
<given-names>R. W.</given-names>
</name>
<name>
<surname>Rowland</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Kimko</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Rostami-Hodjegan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Toward dynamic prescribing information: codevelopment of companion model-informed precision dosing tools in drug development</article-title>. <source>Clin. Pharmacol. drug Dev.</source> <volume>8</volume>, <fpage>418</fpage>&#x2013;<lpage>425</lpage>. <pub-id pub-id-type="doi">10.1002/cpdd.638</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Polasek</surname>
<given-names>T. M.</given-names>
</name>
<name>
<surname>Rostami-Hodjegan</surname>
<given-names>A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Virtual twins: understanding the data required for model-informed precision dosing</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>107</volume>, <fpage>742</fpage>&#x2013;<lpage>745</lpage>. <pub-id pub-id-type="doi">10.1002/cpt.1778</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ribba</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Br&#xe4;m</surname>
<given-names>D. S.</given-names>
</name>
<name>
<surname>Baverel</surname>
<given-names>P. G.</given-names>
</name>
<name>
<surname>Peck</surname>
<given-names>R. W.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Model enhanced reinforcement learning to enable precision dosing: a theoretical case study with dosing of propofol</article-title>. <source>CPT Pharmacometrics Syst. Pharmacol.</source> <volume>11</volume>, <fpage>1497</fpage>&#x2013;<lpage>1510</lpage>. <pub-id pub-id-type="doi">10.1002/psp4.12858</pub-id>
</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ribba</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Dudal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Lav&#xe9;</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Peck</surname>
<given-names>R. W.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Model-informed artificial intelligence: reinforcement learning for precision dosing</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>107</volume>, <fpage>853</fpage>&#x2013;<lpage>857</lpage>. <pub-id pub-id-type="doi">10.1002/cpt.1777</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sheiner</surname>
<given-names>L. B.</given-names>
</name>
<name>
<surname>Beal</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Rosenberg</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Marathe</surname>
<given-names>V. V.</given-names>
</name>
</person-group> (<year>1979</year>). <article-title>Forecasting individual pharmacokinetics</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>26</volume>, <fpage>294</fpage>&#x2013;<lpage>305</lpage>. <pub-id pub-id-type="doi">10.1002/cpt1979263294</pub-id>
</citation>
</ref>
<ref id="B49">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sheiner</surname>
<given-names>L. B.</given-names>
</name>
</person-group> (<year>1969</year>). <article-title>Computer-aided long-term anticoagulation therapy</article-title>. <source>Comput. Biomed. Res.</source> <volume>2</volume>, <fpage>507</fpage>&#x2013;<lpage>518</lpage>. <pub-id pub-id-type="doi">10.1016/0010-4809(69)90030-5</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Sutton</surname>
<given-names>R. S.</given-names>
</name>
<name>
<surname>Barto</surname>
<given-names>A. G.</given-names>
</name>
</person-group> (<year>2018</year>). <source>Reinforcement learning: an introduction</source>. <publisher-name>MIT press</publisher-name>.</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Uster</surname>
<given-names>D. W.</given-names>
</name>
<name>
<surname>Stocker</surname>
<given-names>S. L.</given-names>
</name>
<name>
<surname>Carland</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Brett</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Marriott</surname>
<given-names>D. J.</given-names>
</name>
<name>
<surname>Day</surname>
<given-names>R. O.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>A model averaging/selection approach improves the predictive performance of model-informed precision dosing: vancomycin as a case study</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>109</volume>, <fpage>175</fpage>&#x2013;<lpage>183</lpage>. <pub-id pub-id-type="doi">10.1002/cpt.2065</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Hasselt</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Guez</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Silver</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Deep reinforcement learning with double q-learning</article-title>. <source>Proc. AAAI Conf. Artif. Intell.</source> <volume>30</volume> (<issue>1</issue>). <pub-id pub-id-type="doi">10.1609/aaai.v30i1.10295</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Verhoef</surname>
<given-names>T. I.</given-names>
</name>
<name>
<surname>Ragia</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>de Boer</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Barallon</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Kolovou</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Kolovou</surname>
<given-names>V.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>A randomized trial of genotype-guided dosing of acenocoumarol and phenprocoumon</article-title>. <source>N. Engl. J. Med.</source> <volume>369</volume>, <fpage>2304</fpage>&#x2013;<lpage>2312</lpage>. <pub-id pub-id-type="doi">10.1056/NEJMoa1311388</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>V&#xe9;ronneau-Veilleux</surname>
<given-names>F.</given-names>
</name>
<name>
<surname>Ursino</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Robaey</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>L&#xe9;vesque</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Nekka</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Nonlinear pharmacodynamics of levodopa through Parkinson&#x27;s disease progression</article-title>. <source>Chaos Interdiscip. J. Nonlinear Sci.</source> <volume>30</volume>, <fpage>093146</fpage>. <pub-id pub-id-type="doi">10.1063/5.0014800</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wadelius</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>L. Y.</given-names>
</name>
<name>
<surname>Lindh</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Eriksson</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Ghori</surname>
<given-names>M. J.</given-names>
</name>
<name>
<surname>Bumpstead</surname>
<given-names>S.</given-names>
</name>
<etal/>
</person-group> (<year>2009</year>). <article-title>The largest prospective warfarin-treated cohort supports genetic forecasting</article-title>. <source>Blood, J. Am. Soc. Hematol.</source> <volume>113</volume>, <fpage>784</fpage>&#x2013;<lpage>792</lpage>. <pub-id pub-id-type="doi">10.1182/blood-2008-04-149070</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wadelius</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Pirmohamed</surname>
<given-names>M.</given-names>
</name>
</person-group> (<year>2007</year>). <article-title>Pharmacogenetics of warfarin: current status and future challenges</article-title>. <source>pharmacogenomics J.</source> <volume>7</volume>, <fpage>99</fpage>&#x2013;<lpage>111</lpage>. <pub-id pub-id-type="doi">10.1038/sj.tpj.6500417</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wajima</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Isbister</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Duffull</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>A comprehensive model for the humoral coagulation network in humans</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>86</volume>, <fpage>290</fpage>&#x2013;<lpage>298</lpage>. <pub-id pub-id-type="doi">10.1038/clpt.2009.87</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname>
<given-names>Y.</given-names>
</name>
<name>
<surname>Zhu</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Madabushi</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Liu</surname>
<given-names>Q.</given-names>
</name>
<name>
<surname>Huang</surname>
<given-names>S.-M.</given-names>
</name>
<name>
<surname>Zineh</surname>
<given-names>I.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Model-informed drug development: current us regulatory practice and future considerations</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>105</volume>, <fpage>899</fpage>&#x2013;<lpage>911</lpage>. <pub-id pub-id-type="doi">10.1002/cpt.1363</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wicha</surname>
<given-names>S. G.</given-names>
</name>
<name>
<surname>M&#xe4;rtson</surname>
<given-names>A.-G.</given-names>
</name>
<name>
<surname>Nielsen</surname>
<given-names>E. I.</given-names>
</name>
<name>
<surname>Koch</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Friberg</surname>
<given-names>L. E.</given-names>
</name>
<name>
<surname>Alffenaar</surname>
<given-names>J.-W.</given-names>
</name>
<etal/>
</person-group> (<year>2021</year>). <article-title>From therapeutic drug monitoring to model-informed precision dosing for antibiotics</article-title>. <source>Clin. Pharmacol. Ther.</source> <volume>109</volume>, <fpage>928</fpage>&#x2013;<lpage>941</lpage>. <pub-id pub-id-type="doi">10.1002/cpt.2202</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Xue</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Holford</surname>
<given-names>N.</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2016</year>). <source>Warfarin pkpd: theory, body composition and genotype</source>. <publisher-loc>Lisbon</publisher-loc>: <publisher-name>Annual Meeting of the Population Approach Group in Europe</publisher-name>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zadeh</surname>
<given-names>S. A.</given-names>
</name>
<name>
<surname>Street</surname>
<given-names>W. N.</given-names>
</name>
<name>
<surname>Thomas</surname>
<given-names>B. W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Optimizing warfarin dosing using deep reinforcement learning</article-title>. <source>J. Biomed. Inf.</source> <volume>137</volume>, <fpage>104267</fpage>. <pub-id pub-id-type="doi">10.1016/j.jbi.2022.104267</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Kuang</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Herrero</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Georgiou</surname>
<given-names>P.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>An insulin bolus advisor for type 1 diabetes using deep reinforcement learning</article-title>. <source>Sensors</source> <volume>20</volume>, <fpage>5058</fpage>. <pub-id pub-id-type="doi">10.3390/s20185058</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>