<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Physiol.</journal-id>
<journal-title>Frontiers in Physiology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Physiol.</abbrev-journal-title>
<issn pub-type="epub">1664-042X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fphys.2017.00199</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physiology</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Personalized Medication Response Prediction for Attention-Deficit Hyperactivity Disorder: Learning in the Model Space vs. Learning in the Data Space</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Wong</surname> <given-names>Hin K.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/371323/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Tiffin</surname> <given-names>Paul A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x0002A;</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Chappell</surname> <given-names>Michael J.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/302390/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Nichols</surname> <given-names>Thomas E.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/17823/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Welsh</surname> <given-names>Patrick R.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Doyle</surname> <given-names>Orla M.</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/184481/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lopez-Kolkovska</surname> <given-names>Boryana C.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Inglis</surname> <given-names>Sarah K.</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/408582/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Coghill</surname> <given-names>David</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/424267/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Shen</surname> <given-names>Yuan</given-names></name>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/72448/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ti&#x000F1;o</surname> <given-names>Peter</given-names></name>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/132101/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Warwick Manufacturing Group, Institute of Digital Healthcare, University of Warwick</institution> <country>Coventry, UK</country></aff>
<aff id="aff2"><sup>2</sup><institution>Mental Health and Addiction Research Group, Department of Health Sciences, University of York</institution> <country>York, UK</country></aff>
<aff id="aff3"><sup>3</sup><institution>School of Engineering, University of Warwick</institution> <country>Coventry, UK</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Psychology, Newcastle University</institution> <country>Newcastle upon Tyne, UK</country></aff>
<aff id="aff5"><sup>5</sup><institution>Centre for Neuroimaging Sciences, King&#x00027;s College London</institution> <country>London, UK</country></aff>
<aff id="aff6"><sup>6</sup><institution>Division of Maternal and Child Health Sciences, Ninewells Hospital and Medical School, University of Dundee</institution> <country>Dundee, UK</country></aff>
<aff id="aff7"><sup>7</sup><institution>Departments of Paediatrics and Psychiatry, University of Melbourne</institution> <country>Melbourne, VIC, Australia</country></aff>
<aff id="aff8"><sup>8</sup><institution>School of Computer Science, University of Birmingham</institution> <country>Birmingham, UK</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Zbigniew R. Struzik, University of Tokyo, Japan</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Joaquim Radua, Fidmag Sisters Hospitallers, Spain; Qibin Zhao, RIKEN Brain Science Institute, Japan</p></fn>
<fn fn-type="corresp" id="fn001"><p>&#x0002A;Correspondence: Paul A. Tiffin <email>paul.tiffin&#x00040;york.ac.uk</email></p></fn>
<fn fn-type="other" id="fn002"><p>This article was submitted to Computational Physiology and Medicine, a section of the journal Frontiers in Physiology</p></fn></author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>04</month>
<year>2017</year>
</pub-date>
<pub-date pub-type="collection">
<year>2017</year>
</pub-date>
<volume>8</volume>
<elocation-id>199</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>11</month>
<year>2016</year>
</date>
<date date-type="accepted">
<day>17</day>
<month>03</month>
<year>2017</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2017 Wong, Tiffin, Chappell, Nichols, Welsh, Doyle, Lopez-Kolkovska, Inglis, Coghill, Shen and Ti&#x000F1;o.</copyright-statement>
<copyright-year>2017</copyright-year>
<copyright-holder>Wong, Tiffin, Chappell, Nichols, Welsh, Doyle, Lopez-Kolkovska, Inglis, Coghill, Shen and Ti&#x000F1;o</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) or licensor are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Attention-Deficit Hyperactive Disorder (ADHD) is one of the most common mental health disorders amongst school-aged children with an estimated prevalence of 5% in the global population (American Psychiatric Association, <xref ref-type="bibr" rid="B4">2013</xref>). Stimulants, particularly methylphenidate (MPH), are the first-line option in the treatment of ADHD (Reeves and Schweitzer, <xref ref-type="bibr" rid="B48">2004</xref>; Dopheide and Pliszka, <xref ref-type="bibr" rid="B22">2009</xref>) and are prescribed to an increasing number of children and adolescents in the US and the UK every year (Safer et al., <xref ref-type="bibr" rid="B51">1996</xref>; McCarthy et al., <xref ref-type="bibr" rid="B40">2009</xref>), though recent studies suggest that this is tailing off, e.g., Holden et al. (<xref ref-type="bibr" rid="B33">2013</xref>). Around 70% of children demonstrate a clinically significant treatment response to stimulant medication (Spencer et al., <xref ref-type="bibr" rid="B56">1996</xref>; Schachter et al., <xref ref-type="bibr" rid="B52">2001</xref>; Swanson et al., <xref ref-type="bibr" rid="B61">2001</xref>; Barbaresi et al., <xref ref-type="bibr" rid="B11">2006</xref>). However, it is unclear which patient characteristics may moderate treatment effectiveness. As such, most existing research has focused on investigating univariate or multivariate correlations between a set of patient characteristics and the treatment outcome, with respect to dosage of one or several types of medication. The results of such studies are often contradictory and inconclusive due to a combination of small sample sizes, low-quality data, or a lack of available information on covariates. In this paper, feature extraction techniques such as latent trait analysis were applied to reduce the dimension of on a large dataset of patient characteristics, including the responses to symptom-based questionnaires, developmental health factors, demographic variables such as age and gender, and socioeconomic factors such as parental income. We introduce a Bayesian modeling approach in a &#x0201C;learning in the model space&#x0201D; framework that combines existing knowledge in the literature on factors that may potentially affect treatment response, with constraints imposed by a treatment response model. The model is personalized such that the variability among subjects is accounted for by a set of subject-specific parameters. For remission classification, this approach compares favorably with conventional methods such as support vector machines and mixed effect models on a range of performance measures. For instance, the proposed approach achieved an area under receiver operator characteristic curve of 82&#x02013;84%, compared to 75&#x02013;77% obtained from conventional regression or machine learning (&#x0201C;learning in the data space&#x0201D;) methods.</p>
</abstract>
<kwd-group>
<kwd>attention-deficit hyperactivity disorder</kwd>
<kwd>Bayesian inference</kwd>
<kwd>machine learning</kwd>
<kwd>methylphenidate</kwd>
<kwd>mixed effects model</kwd>
<kwd>personalized medicine</kwd>
<kwd>prognosis</kwd>
<kwd>treatment response</kwd>
</kwd-group>
<contract-num rid="cn001">EP/L000296/1</contract-num>
<contract-sponsor id="cn001">Engineering and Physical Sciences Research Council<named-content content-type="fundref-id">10.13039/501100000266</named-content></contract-sponsor>
<counts>
<fig-count count="9"/>
<table-count count="3"/>
<equation-count count="16"/>
<ref-count count="68"/>
<page-count count="21"/>
<word-count count="14417"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>The ability to predict treatment response (or non-response) in patients with mental health issues is potentially beneficial to both clinicians and patients in a number of ways. First, any treatment is accompanied by the risk of adverse effects&#x02014;where non-response is a probable outcome then the risks of treatment may outweigh the benefits. Second, prediction of treatment response may guide both the dose and choice of medication. For example, where adverse events are dose-dependent then a clinician may chose to abandon a treatment course if a patient was a probable non-responder. Third, response prediction helps to calibrate both clinician and patient expectations of treatment outcomes. Finally, identifying non-responders may prompt a re-appraisal of the diagnosis and formulation of a patient&#x00027;s problem&#x02014;misdiagnosis being one potential cause of non-response. These benefits certainly apply to Attention-Deficit Hyperactive Disorder (ADHD), which is one of the most common developmental disorders among school-aged children with an estimated prevalence of 5% in the general population worldwide (American Psychiatric Association, <xref ref-type="bibr" rid="B4">2013</xref>). Stimulants, particularly methylphenidate (MPH), are the first-line option in the treatment of ADHD (Reeves and Schweitzer, <xref ref-type="bibr" rid="B48">2004</xref>; Dopheide and Pliszka, <xref ref-type="bibr" rid="B22">2009</xref>). Stimulants are prescribed to an increasing number of children and adolescents in the US and the UK every year (Safer et al., <xref ref-type="bibr" rid="B51">1996</xref>; McCarthy et al., <xref ref-type="bibr" rid="B40">2009</xref>), though recent studies suggest that this trend is tailing off e.g., Holden et al. (<xref ref-type="bibr" rid="B33">2013</xref>). The beneficial effects of stimulant medication on the core symptoms of ADHD have been demonstrated by numerous clinical trials, reviews and meta-analyses (Banaschewski et al., <xref ref-type="bibr" rid="B10">2006</xref>; Greenhill et al., <xref ref-type="bibr" rid="B29">2006</xref>; van der Oord et al., <xref ref-type="bibr" rid="B65">2008</xref>; Storeb&#x000F8; et al., <xref ref-type="bibr" rid="B59">2015</xref>). Nevertheless, adverse effects of the medications are also common (Storeb&#x000F8; et al., <xref ref-type="bibr" rid="B59">2015</xref>). The findings from previous research suggest that around 70% of children demonstrate a clinically significant treatment response to stimulant medication (Spencer et al., <xref ref-type="bibr" rid="B56">1996</xref>; Schachter et al., <xref ref-type="bibr" rid="B52">2001</xref>; Swanson et al., <xref ref-type="bibr" rid="B61">2001</xref>; Barbaresi et al., <xref ref-type="bibr" rid="B11">2006</xref>). However, it is unclear which patient characteristics may moderate treatment effectiveness and whether non-response can be predicted.</p>
<p>To date, achieving accurate predictions of the clinical outcomes for patients with ADHD has proven elusive&#x02014;most of the literature has focused on investigating the potential correlations between a set of patient characteristics and the outcome following treatment with one or more types of medication. Information relating to patient characteristics has mostly been in the form of subjective questionnaire ratings, clinical notes and qualitative psychometric data; for example, the ratings from symptom-based questionnaires such as the Swanson, Nolan, and Pelham (SNAP) questionnaire (Swanson et al., <xref ref-type="bibr" rid="B62">1983</xref>; Atkins et al., <xref ref-type="bibr" rid="B9">1985</xref>; Swanson, <xref ref-type="bibr" rid="B60">1992</xref>; Bussing et al., <xref ref-type="bibr" rid="B18">2008</xref>), along with demographic variables such as age, sex and social economic background. The results from such studies are often contradictory and inconclusive due to small sample sizes and/or limited availability and quality of data, especially in the temporal (longitudinal) domain.</p>
<p>Along with more conventional statistical approaches, machine learning has also shown promise in predicting treatment response or prognosis in healthcare applications. Indeed, recently a random forest regression analysis was used to predict outcome in a group of patients affected by Obsessive Compulsive Disorder (OCD) from a relatively small pool of questionnaire items, with a reported error rate of 24.6% (Askland et al., <xref ref-type="bibr" rid="B7">2015</xref>). Likewise, there has been a previous attempt to use machine learning techniques to predict treatment response in ADHD (Kim et al., <xref ref-type="bibr" rid="B36">2015</xref>); support vector machine classification from this study was reported as 84.6% accurate (not to be confused with the balanced accuracy measure used in this paper). However, in addition to demographic and clinical questionnaire-derived data, the study used genetic as well as neuroimaging and neuropsychological information as inputs. Such data are unlikely to be readily available to clinicians in routine practice.</p>
<p>In this paper we investigate whether the inclusion of prior knowledge relating to the potential mechanism behind the presentation of a mental health condition and characteristics of individual patients can add value in predicting treatment. Thus, we hypothesized that a pragmatic machine learning approach based on a mechanistic or parametric model (a &#x0201C;learning in the model space&#x0201D; framework) for treatment response prediction may offer an advantage over more conventional methods (Brodersen et al., <xref ref-type="bibr" rid="B16">2011</xref>; Doyle et al., <xref ref-type="bibr" rid="B23">2013</xref>; Chen et al., <xref ref-type="bibr" rid="B20">2014</xref>; Shen et al., <xref ref-type="bibr" rid="B55">2016</xref>). This method represents each newly observed patient through a model; the models are personalized such that individual differences are accounted for by a set of subject-specific parameters. In the case of ADHD, developing a plausible mechanistic model is not straightforward&#x02014;despite decades of research, the underlying mechanism for the disorder is not well understood. In addition, any mechanistic model would have to be based on data that are likely to be available in good, but routine, clinical practice.</p>
<p>This paper documents, within the &#x0201C;learning in the model space&#x0201D; framework, a Bayesian linear regression model for the prediction of treatment response in a cohort of children diagnosed and treated for ADHD in the UK. The performance of this new approach is then compared with conventional regression and machine learning methods (&#x0201C;learning in the data space&#x0201D;) to assess whether or not the new approach offers benefits, and if so under what circumstances.</p>
</sec>
<sec sec-type="materials and methods" id="s2">
<title>2. Materials and methods</title>
<sec>
<title>2.1. Participants</title>
<p>The children enrolled in the study were drawn from the ADHD Drug Use and Chronic Effects (ADDUCE) cohort study (The ADDUCE Consortium, <xref ref-type="bibr" rid="B63">2016</xref>), covered by a data sharing agreement with patient consent. The participants were from the UK NHS Tayside region who had attended the ADHD treatment clinics held at Dundee and Perth, UK. 262 families of eligible children were contacted, of which 181 (70%) were recruited and data on 173 of them were obtained for the purpose of this study. In addition, data were available on 94 healthy controls. Out of the 173 patients (whose baseline data are available), 157 of them started dose optimization studies and therefore longitudinal (temporal) data are available (See Section 3.1). To be eligible for the ADHD group, children had to be 6&#x02013;17 years of age, have a clinical diagnosis of ADHD (see below), have had no previous medical history of methylphenidate use (medication-na&#x000EF;ve) and have parental and child consent/assent to commence. The criteria for the healthy control group were similar apart from them having no current or previous psychiatric diagnoses. The recruitment was carried out over a 30-month period from January 2012&#x02013;August 2014.</p>
<p>All patients in the ADHD group had already been clinically diagnosed with ADHD; this diagnosis was based on the clinical judgment of the assessing physician, informed by structured interviews with parents/carers, information provided by the child&#x00027;s school, direct observation of the child at the clinic, and at times, in their educational setting. Thus, the physician had to be satisfied that the child fulfilled the diagnostic criteria for a hyperkinetic disorder according to the International Classification of Diseases 10th edition (ICD-10) (World Health Organization, <xref ref-type="bibr" rid="B67">2010</xref>), or ADHD as defined by the Diagnostic and Statistical Manual 4th edition (DSM-IV) (American Psychiatric Association, <xref ref-type="bibr" rid="B3">2000</xref>). This means that the child had to demonstrate disabling and pervasive inattentiveness, hyperactivity, and impulsivity across a range of settings. The clinic was designed to implement a &#x0201C;dose optimization titration&#x0201D; scheme of medication in children diagnosed with ADHD. This involved giving increasing doses of methylphenidate (as the first line medication) at roughly weekly intervals until remission from symptoms was achieved or problematic adverse effects were encountered. If remission was not achieved with a first line medication within recommended dosage limits, or if problematic side-effects were encountered then a second line drug was initiated, and again, increased in dosage, as before.</p>
</sec>
<sec>
<title>2.2. Assessment</title>
<p>A range of baseline social and demographic factors was recorded, including parental marital status, family composition, and socioeconomic status as indicated by the Scottish Index of Multiple Deprivation (SIMD) 2012 (APS Group Scotland, <xref ref-type="bibr" rid="B6">2012</xref>) derived from the family home postcode. A history of any previous psychiatric or non-psychiatric medication exposure was recorded, as were any physical health issues. Verbal and non-verbal intellectual functioning was estimated from parental reports and any educational issues noted. Problems with anxiety and low mood were rated using the short form of the Mood and Feelings Questionnaire (MFQ) with the parents, and where appropriate, the child as informants (Angold et al., <xref ref-type="bibr" rid="B5">1996</xref>). Dystonia and abnormal movements were recorded using the Abnormal Involuntary Movement Scale (AIMS) (Guy, <xref ref-type="bibr" rid="B30">1974</xref>, pp. 534&#x02013;537). Oppositional and ADHD symptoms and behaviors were rated, according to parental report, using the Swanson, Nolan, and Pelham (SNAP-IV) questionnaire (Swanson et al., <xref ref-type="bibr" rid="B62">1983</xref>). Any substance used by the participants was recorded using the Substance Use Questionnaire (SUQ). Fine motor issues were recorded using the Developmental Coordination Disorder Questionnaire 2007 (DCDQ&#x00027;07). Several sections of the Development and Well-Being Assessment (DAWBA) were used (Goodman et al., <xref ref-type="bibr" rid="B27">2000</xref>); these were (1) Rapidly Changing Mood (child and parent versions), (2) Tic disorders, including the Tourette syndrome, (3) Awkward and troublesome behavior. Tic severity (where present) was also rated using the Yale Global Tic Severity Scale (YGTSS) (Leckman et al., <xref ref-type="bibr" rid="B38">1989</xref>). Possible behaviors associated with an underlying Autism Spectrum Disorder (ASD) were evaluated using the Social Communication Questionnaire (SCQ) (Rutter et al., <xref ref-type="bibr" rid="B50">2003</xref>). The Strengths and Difficulties Questionnaire (SDQ) (Goodman, <xref ref-type="bibr" rid="B26">1997-07</xref>) was used to rate parental perceived levels of pro-social behavior, hyperactivity/impulsivity, conduct problems, emotional symptoms and peer relationship problems. The overall clinical impression was recorded using the Clinical Global Impression&#x02014;Severity scale (CGI-S) (Guy, <xref ref-type="bibr" rid="B30">1974</xref>, pp. 218&#x02013;222) and Children&#x00027;s Global Assessment Scale (CGAS) (Shaffer et al., <xref ref-type="bibr" rid="B54">1983</xref>).</p>
<p>Responses to medication, in terms of levels of ADHD symptoms, were reported by parents and recorded using the SNAP-IV questionnaire at each visit. Likewise, any potential adverse effects and co-morbidity problems were reported using the standard clinic proforma, along with weight, height and blood pressure of the child at each visit.</p>
</sec>
<sec>
<title>2.3. Feature extraction/factor analysis</title>
<p>The aforementioned questionnaires included a large number of items with categorical (binary or ordinal) response formats. Thus, in order to facilitate model development by reducing the dimensionality of the data whilst minimizing the loss of information, a series of factor (latent trait) analyses were conducted.</p>
<p>The key questionnaires used in the modeling process were the SCQ, the SDQ, and the SNAP-IV (see the previous section). In particular, the SNAP-IV scores served as the outcome variables, which indicated whether symptomatic remission had been achieved, following the dose-optimized titration of medication. The factor analyses sought to identify the dimensionality underlying the responses to the questionnaires and, consequently, the standardized factor scores represented the level of trait for each patient in that underlying dimension or construct.</p>
<p>In order to estimate the dimensionality, the sample of 173 patients and 94 healthy controls was randomly divided into two roughly equal exploratory and confirmatory datasets. A parallel analysis (Horn, <xref ref-type="bibr" rid="B34">1965</xref>), adapted for categorical data, was then implemented in the freeware FACTOR (Lorenzo-Seva and Ferrando, <xref ref-type="bibr" rid="B39">2006</xref>) using unweighted least squares (ULS) estimation method. A weighted &#x0201C;promax&#x0201D; rotation was deployed to achieve factor simplicity (Abdi, <xref ref-type="bibr" rid="B1">2003</xref>). The maximum number of plausible factors (latent variables) was assumed to be indicated at the point where the eigenvalues of the factors in randomly generated data exceeded those observed in the real data. A series of exploratory factor analyses (EFAs&#x02013;adapted for categorical dependent variables) were then conducted to aid interpretation of the factors. Oblique &#x0201C;geomin&#x0201D; rotation was used (Asparouhov and Muth&#x000E9;n, <xref ref-type="bibr" rid="B8">2009</xref>), assuming that, as in almost all psychological measures, underlying latent traits would be correlated with each other to some extent (Thurstone, <xref ref-type="bibr" rid="B64">1931</xref>). A series of confirmatory factor analyses (CFAs) were then conducted using the held-back, confirmatory data (see Section 3.1 on cross-validation), in order to ensure that the factor structures derived fitted the data adequately. All EFAs and CFAs were conducted in the Mplus software version 7.1, using robust weighted least squares with mean and variance adjustment (WLSMV) as the estimation method (Muth&#x000E9;n et al., <xref ref-type="bibr" rid="B41">1997</xref>). Remission was defined by a child having a reported factor score in the hyperactive and inattentive dimensions (both elicited from factor analysis) equivalent to a mean item score in the SNAP-IV of one or less, which is conventionally taken to indicate symptomatic remission (Hechtman, <xref ref-type="bibr" rid="B32">2005</xref>; Chou et al., <xref ref-type="bibr" rid="B21">2012</xref>). The resulting symptom score thresholds are only slightly different for inattentiveness and hyperactivity (&#x02212;0.97 vs. &#x02212;0.92).</p>
</sec>
</sec>
<sec id="s3">
<title>3. Modeling approach</title>
<p>The causal factor model, shown in Figure <xref ref-type="fig" rid="F1">1</xref>, was derived using a rapid review approach to appraise and synthesize the existing evidence (Khangura et al., <xref ref-type="bibr" rid="B35">2012</xref>). This model also took into account the nature of the data available in the cohort and was modified accordingly. The goal is not for the causal model to be comprehensive or definitive, but to identify from the literature as many potential factors relating to treatment response as there are available from the dataset, as well as helping to elicit the Bayesian prior distributions (Section 3.2.1). Model development was based on a literature review. This involved running searches in the EMBASE, MEDLINE and PsycINFO databases using the synonyms for ADHD (e.g., hyperkinesis) combined with terms relating to treatment outcome or response, and the names of the medications (both scientific and trademarks, full and abbreviated) prescribed in the cohort. The medications include 1) immediate release methylphenidate (IR-MPH, e.g., Ritalin&#x000AE;), 2) long-acting methylphenidate (XR-MPH, e.g., Concerta XL&#x000AE;, Equasym XL&#x000AE;, Medikinet XL&#x000AE;), 3) dextroamphetamine (DEX, e.g., Dexedrine&#x000AE;) including its prodrug lisdexamfetamine dimesylate (e.g., Elvanse&#x000AE;), and 4) atomoxetine (ATOM, e.g., Strattera&#x000AE;). Secondary sources were followed up. The quality of trial-based studies could be appraised using the CONSORT checklist (Schulz et al., <xref ref-type="bibr" rid="B53">2010</xref>) and observational studies via the STROBE guidance (von Elm et al., <xref ref-type="bibr" rid="B66">2007</xref>). Two of the authors (HKW and PAT) then made a judgment, based on the findings reported in the literature and the perceived likelihood of bias or uncertainty as to what extent variables in the model might be related to treatment response in ADHD. The model derived was consequently used to populate prior distributions for the patient-specific model parameters (i.e., the hyperpriors). Where the evidence was uncertain or inconsistent, the variances (i.e., imprecision) of the hyperpriors were increased.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p><bold>High level causal factor model of treatment response prediction in ADHD</bold>.</p></caption>
<graphic xlink:href="fphys-08-00199-g0001.tif"/>
</fig>
<p>Not every piece of information mentioned in Section 2.2 was used for the purpose of modeling, because of insufficient data or multicollinearity between the variables. The causal factor model was then simplified based on the breadth of available data from the cohort, leading to a much reduced model as shown in Figure <xref ref-type="fig" rid="F2">2</xref>. Some factors were combined through another layer of feature extraction; for example, the <italic>motor</italic> and <italic>control</italic> latent factors, themselves also obtained from applying feature extraction to the DCDQ&#x00027;07 questionnaire data (see Section 2.2), were combined with the <italic>non-verbal communication</italic> factor from the SCQ questionnaire to obtain a developmental adversity factor. Some factors were not obtained from standard questionnaires; for example, the <italic>perinatal adversity</italic> factor (see Figure <xref ref-type="fig" rid="F2">2</xref>) was constructed from birth weight and gestation age; the <italic>family size and socioeconomic status</italic> factor combined the number of siblings, parental house ownership (owned, mortgaged or rented) and the SIMD 2012 index (APS Group Scotland, <xref ref-type="bibr" rid="B6">2012</xref>).</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p><bold>Reduced causal factor model of treatment response prediction in ADHD</bold>.</p></caption>
<graphic xlink:href="fphys-08-00199-g0002.tif"/>
</fig>
<sec>
<title>3.1. Data</title>
<p>There were 267 subjects whose baseline characteristics were measured (173 clinically diagnosed with ADHD and 94 healthy controls) at the first clinical appointment. Of the 173 non-controls, 157 were enrolled in dose optimization titration studies with parental consent, for whom longitudinal (temporal) data are available. The 157 patients with longitudinal data were randomized and 10-fold cross-validation partitions were constructed. Subjects were partitioned into 10 subgroups of roughly equal size in a patient-coherent fashion, i.e., data from a single patient only appeared in a single fold.</p>
<p>For all models investigated in this paper, a single fold was used as the validation dataset and the remaining nine folds were combined to serve as the training dataset. This process was iterated until each fold had served as validation data exactly once.</p>
<sec>
<title>3.1.1. Baseline characteristics</title>
<p>We labeled the patient subjects by the indexing variable <italic>s</italic> &#x0003D; 1, 2, &#x02026;, <italic>N</italic>. A set of <italic>L</italic> patient-specific baseline continuous latent factors, encoded in a row vector <bold>b</bold><sub><italic>s</italic></sub> &#x02208; &#x0211D;<sub>1&#x000D7;<italic>L</italic></sub> was obtained by performing feature extraction as described in Section 2.3 over the questionnaires detailed in Section 2.2. Referring to Figure <xref ref-type="fig" rid="F2">2</xref>, <italic>L</italic> &#x0003D; 14 factors were used for the baseline. Data from the controls in addition to the training dataset were utilized during feature extraction to ensure that the resulting latent factor models can sufficiently encompass the entire range of characteristics from ADHD patients to normal children. The resulting continuous latent factors would, in theory, be sufficiently representative of the information conveyed by the categorical questionnaire response variables.</p>
<p>To ensure that validation data were strictly not used for the model building, feature extraction was first performed using only training data from each of the folds (plus all the controls). This resulted in 10 sets of factor scores corresponding to each fold. The factor model structures (e.g., the number of factors per questionnaire) over the folds did not change across the folds, as statistical fit indices and Chi-square difference tests did not suggest that any changes were necessary. The factor models were then used to estimate the baseline factor scores for the validation sets in each of the folds.</p>
<p>Each of the 10 cross-validation runs resulted in a set of corresponding continuous latent factors, which were used as inputs to subsequent models. The models were trained and validated using the same training-validation partitioning used in the feature extraction process.</p>
</sec>
<sec>
<title>3.1.2. Longitudinal data</title>
<p>Each of the 157 subjects with longitudinal data visited the clinic a varying number of times&#x02014;from titration, stabilization to continuing care; the number of doctor&#x00027;s appointments, <italic>A</italic><sub><italic>s</italic></sub>, varies from 1 to 22. At each appointment, the parent or guardian of the patient was asked to fill in an 18-item SNAP-IV questionnaire, which measures the degree of inattentiveness and hyperactivity. The responses were entered into a factor model (identified through feature extraction) to extract a continuous symptom score for inattentiveness and hyperactivity. We denote the appointment number by the indexing variable <italic>a</italic> so that <italic>a</italic> &#x0003D; 1, 2, &#x02026;, <italic>A</italic><sub><italic>s</italic></sub>. Let the independent &#x0201C;input&#x0201D; variables <italic>m</italic><sub><italic>a</italic>, 1</sub>, <italic>m</italic><sub><italic>a</italic>, 2</sub>, <italic>m</italic><sub><italic>a</italic>, 3</sub>, <italic>m</italic><sub><italic>a</italic>, 4</sub> be the four types of medications, respectively, IR-MPH, XR-MPH, DEX, and ATOM for subject <italic>s</italic> at appointment <italic>a</italic>. Using datasheets for the medicines used, the dosages of DEX and ATOM were normalized to an equivalent daily dosage (EDD) of IR-MPH. For all <italic>a</italic> and <italic>s</italic>, this results in input and output matrices of the form:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Input</mml:mtext><mml:mo>:</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>M</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Output</mml:mtext><mml:mo>:</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>&#x02026;</mml:mo></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x022BA;</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>r</italic> is the symptom severity measure and can be either the inattentiveness factor score or the hyperactivity factor score.</p>
<p>The combined EDDs of medications (for the 4 types) prescribed over the appointment number for all patients are plotted as a boxplot in Figure <xref ref-type="fig" rid="F3">3</xref>. One can observe that as forced titration progressed over the appointments, the overall dosage level increased.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p><bold>Boxplot of combined equivalent (in IR-MPHunits) daily dosages of medications taken for all patients vs. appointment number</bold>. Red horizontal lines: median; boxes: interquartile range; whiskers: 95% confidence intervals; red crosses: outliers.</p></caption>
<graphic xlink:href="fphys-08-00199-g0003.tif"/>
</fig>
<p>Figure <xref ref-type="fig" rid="F4">4</xref> shows the distribution of inattentiveness and hyperactivity symptom factor scores for the patients for each appointment. The lower the factor scores, the less severe the symptoms are. In terms of a general trend, one can clearly see an effective and quick reduction in symptom levels over the first 5 appointments, as stimulant medication prescription ramps up during forced titration. The symptom scores cease to improve for appointments 6&#x02013;8, after which a slight increase can be observed. This hints at adherence or persistence issues, but the available data do not allow further investigation&#x02014;as such issues are not consistently reported by the parent/guardian or recorded in the clinical notes. While the model has no mechanism for modeling such effects, the adaptive learning nature of the Bayesian algorithm is able to self-correct and compensate for small deviations, for example, by &#x0201C;learning&#x0201D; to weight down the dose-response parameter for a given medication when the patient has a low adherence.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p><bold>Boxplots of symptom scores across all patients vs. appointment number</bold>. <bold>(A)</bold> Inattentiveness, <bold>(B)</bold> hyperactivity. Red horizontal lines: median; boxes: interquartile range; whiskers: 95% confidence intervals; red crosses: outliers.</p></caption>
<graphic xlink:href="fphys-08-00199-g0004.tif"/>
</fig>
</sec>
</sec>
<sec>
<title>3.2. Treatment response model formulation</title>
<p>The treatment outcome is modeled as a linear combination of the baseline variables and the medication dosage,</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003C9;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mo>&#x003F5;</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>X</mml:mi></mml:mstyle><mml:mi>s</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>b</mml:mi></mml:mstyle><mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mo>&#x000A0;</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd><mml:mtd><mml:mrow><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>M</mml:mi></mml:mstyle><mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mo>&#x022EE;</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:msub><mml:mstyle mathvariant='bold' mathsize='normal'><mml:mi>b</mml:mi></mml:mstyle><mml:mi>s</mml:mi></mml:msub></mml:mrow></mml:mtd><mml:mtd><mml:mrow><mml:mo>&#x000A0;</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><inline-formula><mml:math id="M5"><mml:mo>&#x003F5;</mml:mo><mml:mo>&#x0007E;</mml:mo><mml:mi mathvariant="-tex-caligraphic">N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x003C3;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x003F5;</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mo>&#x0211D;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is an error term and &#x003C9;<sub><italic>s</italic></sub> &#x02208; &#x0211D;<sub><italic>P</italic>&#x000D7;1</sub> is the <italic>subject-specific</italic> parameter vector moderating the effect of the baseline variables <bold>b</bold><sub><italic>s</italic></sub> on the treatment response, i.e., it accounts for how large an effect each of the various baseline variables or medication types has on treatment outcome. &#x0201C;Subject-specific&#x0201D; means that the parameter vector was allowed to be different for each subject so that patients with similar baseline characteristics can still have a different prediction outcome. The number of free parameters required is <italic>P</italic> &#x0003D; <italic>L</italic> &#x0002B; 4 &#x0002B; 1 &#x0003D; 19 for <italic>L</italic> &#x0003D; 14 (see Figure <xref ref-type="fig" rid="F2">2</xref>).</p>
<p>The baseline variables remain unchanged over different appointments while the medication dosage may vary according to the titration regime specified by the clinician. Hence, every row of the matrix <bold>X</bold><sub><italic>s</italic></sub> contains the same baseline characteristic vector for an individual patient, combined with the medication dosage vector. The row number in <bold>X</bold><sub><italic>s</italic></sub> corresponds to the appointment number.</p>
<p>Because the number of visits <italic>A</italic><sub><italic>s</italic></sub> of the subjects was usually fewer than the dimension of the parameter space <italic>P</italic>, the problem is mathematically underdetermined. Hence, classical least squares regression methods would fail without an appropriate regularization (Goodfellow et al., <xref ref-type="bibr" rid="B28">2016</xref>). To this end, we employ a Bayesian formulation. In particular, the Bayesian linear regression was used to model the temporal evolution of the dose-response relationship for each patient.</p>
<p>In essence, a Bayesian approach allows prior or expert knowledge to be encoded into the problem formulation and this enables a probabilistic solution to be found despite the limited data available.</p>
<sec>
<title>3.2.1. Prior distributions and knowledge</title>
<p>The joint prior probability density function Pr(&#x003C9;, &#x003C3;<sup>2</sup>) is given in Equation (A3, <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>). In this exercise, the causal treatment response model based on the literature was used to constrain the prior of &#x003C9; and its covariance matrix <inline-formula><mml:math id="M6"><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>.</p>
<p>When eliciting the prior, quantitative information from the literature was not used, e.g., setting the mean of the prior distribution of the parameter vector omega to a specific numerical value. This is because the demography, sample sizes and effect sizes across literature vary and there is no correct way to normalize them. Instead, only the sign (direction) of the effect was encoded. For example, there is evidence that methylphenidate improves treatment response in the literature (a positive dose results in lower symptom score), therefore a negative value of &#x02212;1 was specified for the columns of &#x003C4;<sub>0</sub> corresponding to <italic>m</italic><sub><italic>a</italic>, 1</sub>&#x02026;<italic>m</italic><sub><italic>a</italic>, 4</sub> in Equation (1). For positive associations with symptom scores, &#x0002B;1 was used instead. The same magnitude is used in other factors (i.e., either &#x0002B;1 or &#x02212;1).</p>
<p>Each cohort study or clinical trial from the rapid review was appraised, respectively, using the STROBE and the CONSORT checklists by counting the number of pass and fail items out of the total. Evidence from the literature was marked as good quality when both the checklist score was similar to other studies on the same topic (within 20% from the best) and effect sizes were statistically significant as reported by the authors for the sample size used. The diagonals of the covariance matrix <inline-formula><mml:math id="M7"><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> were assigned an initial value of one; for every contradicting evidence (the effect sizes are opposite in direction) satisfying these criteria, 0.5 was added to the corresponding variance in the covariance matrix. A higher value may be specified if necessary, to ensure that the prior distribution of parameter omega spans both positive and negative sides sufficiently&#x02014;within one standard deviation of &#x003C4;<sub>0</sub>. On the other hand, if the effect sizes are positive the variance was reduced by 0.1 for each supporting studies, at the same time ensuring the variance does not go below 0.5. The (lack of) proposed existence of causal links between the variables in the causal model (see Figure <xref ref-type="fig" rid="F1">1</xref>) ensures the sparsity of the covariance matrix <inline-formula><mml:math id="M8"><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>.</p>
<p>While these numbers may not be completely objective, the amount of data available means that the sensitivity of the results to the prior is low&#x02014;sensitivity analysis shows that the effect of scaling the prior covariance between 50 and 150% of its original values changes the errors by about 5% of the training root mean squared (rms) error, and 3% for the validation rms error.</p>
</sec>
<sec>
<title>3.2.2. Posterior distributions</title>
<p>The posterior distribution is given in Equation (A4, <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>), where the parameters of the distributions are obtained through Equation (A5) in <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>. In the training set, Bayesian learning uses data from all of the appointments that a subject had, in which case <italic>n</italic> &#x0003D; <italic>A</italic><sub><italic>s</italic></sub> where, as before, <italic>A</italic><sub><italic>s</italic></sub> is the total number of visits or appointments a subject has and data are available for. We introduce the simplified notations after Bayesian update has been applied to the training set using Equation (A5) in <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>, so that</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x02192;</mml:mo><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:msub><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>&#x02192;</mml:mo><mml:mover accent="true"><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:msub><mml:mrow><mml:mo>&#x003B1;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>&#x02192;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B1;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mtext class="textrm" mathvariant="normal">&#x000A0;and&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mo>&#x003B2;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mo>&#x02192;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B2;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This notation will be used in later sections. The reader should be reminded that the parameters are derived from each subject and hence are different across subjects. Notice that in a prediction exercise (instead of retrospective regression formulated here), the learning can be applied incrementally for each future observation with each update using just the new observation.</p>
</sec>
</sec>
<sec>
<title>3.3. Virtual patient profile</title>
<p>When a new patient (denoted by <italic>s</italic> &#x0003D; &#x0002A;) is received, one can measure their baseline variables <bold>b</bold><sub>&#x0002A;</sub>, but not their model parameter space &#x003C9;<sub>&#x0002A;</sub>. The goal is to estimate a <italic>virtual patient profile</italic> that is believed to best describe the new patient using only the available baseline measurements. To do this one derives the mathematical mapping functions from the baseline characteristics of a patient to their posterior parameters <bold>b</bold><sub>&#x0002A;</sub> &#x021A6; Pr(&#x003C9;<sub>&#x0002A;</sub>) and <inline-formula><mml:math id="M10"><mml:mrow><mml:mo class="qopname">Pr</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x003C3;</mml:mo></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>, such that a prediction can be made from the baseline variables. These functions are forged using machine learning on the existing pool of training data. Since Pr(&#x003C9;<sub><italic>s</italic></sub>) is parameterized by <inline-formula><mml:math id="M11"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> and likewise <inline-formula><mml:math id="M12"><mml:mo class="qopname">Pr</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mo>&#x003C3;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> by (&#x003B1;<sub><italic>s</italic></sub>, &#x003B2;<sub><italic>s</italic></sub>), one has to learn the mappings from the baseline variables to the parameters. The learnt mathematical mapping functions can then be used to obtain estimates of <inline-formula><mml:math id="M13"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M14"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B1;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B2;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, which represent the virtual patent profile for the new patient in the model space.</p>
<p>Due to the conjugate nature of the priors, one does not need to derive the hyperparameters <inline-formula><mml:math id="M15"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mo>&#x003B1;</mml:mo><mml:mo>^</mml:mo></mml:mover><mml:mo>*</mml:mo></mml:msub></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M16"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mo>&#x003B2;</mml:mo><mml:mo>^</mml:mo></mml:mover><mml:mo>*</mml:mo></mml:msub></mml:mrow></mml:math></inline-formula> from Equation (5) for the purpose of having point estimates for the treatment response prediction. However, these hyperparameters are necessary in order to derive the posterior distribution of the predicted value&#x02014;commonly referred to as the posterior predictive distribution. Knowing the distribution allows us to approximate the uncertainties of the estimates, e.g., 95% confidence intervals. Two methods were proposed for learning the mappings from the baseline variables to the virtual patient profile and they are introduced in the following subsections.</p>
<sec>
<title>3.3.1. Method 1: generalized linear regression</title>
<p>To determine the mappings, one finds the functions: a) <inline-formula><mml:math id="M17"><mml:msub><mml:mrow><mml:mstyle class="text"><mml:mtext class="textsl" mathvariant="italic">f</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02248;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula> and b) <inline-formula><mml:math id="M18"><mml:msub><mml:mrow><mml:mstyle class="text"><mml:mtext class="textsl" mathvariant="italic">f</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>u</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02248;</mml:mo><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> where <bold>u</bold><sub><italic>s</italic></sub> is a row vector containing non-zero elements of the upper (or lower) triangular part of <inline-formula><mml:math id="M19"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>. Since <inline-formula><mml:math id="M20"><mml:msup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> is a covariance matrix (hence symmetric), knowledge of the lower/upper half of the off-diagonal elements plus the diagonal elements is sufficient to fully recreate the matrix.</p>
<p>The mappings are learnt from the training data, in which the posterior distributions of <inline-formula><mml:math id="M21"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M22"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> are already available through Equation (A5) in <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>. Linear regression models of the form <inline-formula><mml:math id="M23"><mml:mstyle mathvariant="bold"><mml:mtext>Y</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>P</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mstyle mathvariant="bold"><mml:mtext>B</mml:mtext></mml:mstyle></mml:math></inline-formula> were used to model the two mappings, where the matrix <bold>B</bold> has rows of <bold>b</bold><sub><italic>s</italic></sub> vectors&#x02014;one for each subject in the training set&#x02014;and similarly <bold>Y</bold> is composed of rows of a) &#x003C4;<sub><italic>s</italic></sub> for determining <italic>f</italic><sub>&#x003C4;</sub> or b) <bold>u</bold><sub><italic>s</italic></sub> for determining <italic>f</italic><sub><bold>u</bold></sub>. The least squares solutions for the models are given by the Moore-Penrose pseudo-inverse,</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Q</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>B</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022BA;</mml:mo></mml:mrow></mml:msup><mml:mstyle mathvariant="bold"><mml:mtext>B</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>B</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>&#x022BA;</mml:mo></mml:mrow></mml:msup><mml:mstyle mathvariant="bold"><mml:mtext>Y</mml:mtext></mml:mstyle><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>For prediction, <italic>f</italic><sub>&#x003C4;</sub> and <italic>f</italic><sub><bold>u</bold></sub> can both be formulated as <inline-formula><mml:math id="M25"><mml:mstyle class="text"><mml:mtext class="textsl" mathvariant="italic">f</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>Q</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:math></inline-formula>.</p>
<p>The posterior estimates for the hyperparameters for a new patient are taken as the averaged values of <inline-formula><mml:math id="M26"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B1;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M27"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B2;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> across all subjects in the training set, resulting in <inline-formula><mml:math id="M28"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mo>&#x003B1;</mml:mo><mml:mo>^</mml:mo></mml:mover><mml:mo>*</mml:mo></mml:msub><mml:mo>=</mml:mo><mml:mn>5</mml:mn><mml:mo>.</mml:mo><mml:mn>5</mml:mn></mml:mrow></mml:math></inline-formula> and <inline-formula><mml:math id="M29"><mml:mrow><mml:msub><mml:mover accent="true"><mml:mo>&#x003B2;</mml:mo><mml:mo>^</mml:mo></mml:mover><mml:mo>*</mml:mo></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>.</mml:mo><mml:mn>7</mml:mn></mml:mrow></mml:math></inline-formula>.</p>
</sec>
<sec>
<title>3.3.2. Method 2: Gaussian kernel weighted averaging</title>
<p>An alternative method is to find <inline-formula><mml:math id="M30"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M31"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> using a weighted average of <inline-formula><mml:math id="M32"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02032;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M33"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02032;</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, with <inline-formula><mml:math id="M34"><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02032;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D54A;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula> being a subset of subjects in the existing training pool whose baseline variables (<inline-formula><mml:math id="M35"><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02032;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula>) were &#x0201C;similar&#x0201D; to those of the new patient (<bold>b</bold><sub>&#x0002A;</sub>). Highly similar subjects will have a higher influence on the value of <inline-formula><mml:math id="M36"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M37"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>. The &#x0201C;(dis)similarity&#x0201D; <italic>d</italic><sub><italic>s</italic></sub> is measured using the pairwise euclidean distance between <bold>b</bold><sub><italic>s</italic></sub> and <bold>b</bold><sub>&#x0002A;</sub>, such that</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M38"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>b</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mo>&#x022BA;</mml:mo></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This is then sorted and the 17.5% of subjects in <italic>&#x1D54A;</italic><sub>&#x0002A;</sub> with the smallest &#x0201C;dissimilarity&#x0201D; values are kept; this percentage value was chosen as it resulted in the lowest validation error. The weighting <italic>w</italic><sub><italic>s</italic></sub> was taken as the normalized Gaussian kernel</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M39"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mo>&#x003BB;</mml:mo><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02032;</mml:mo></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D54A;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mo>&#x003BB;</mml:mo><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>&#x02032;</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the parameter value &#x003BB; &#x0003D; 1.15 was chosen as it again resulted in the lowest validation error. Using Equations (7) and (8), one can estimate <inline-formula><mml:math id="M40"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M41"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> as</p>
<disp-formula id="E9"><label>(9a)</label><mml:math id="M42"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:mi>s</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D54A;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E10"><label>(9b)</label><mml:math id="M58"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:mi>s</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D54A;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>and similarly the estimates of the hyperparameters are calculated using</p>
<disp-formula id="E16"><label>(9c)</label><mml:math id="M43"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B1;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:mi>s</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D54A;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003B1;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E11"><label>(9d)</label><mml:math id="M44"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B2;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:mi>s</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x1D54A;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003B2;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
</sec>
</sec>
<sec>
<title>3.4. Prediction using the posterior predictive distribution</title>
<p>When a new subject visits the clinician, their <bold>b</bold><sub>&#x0002A;</sub> vector may be measured and used to approximate <inline-formula><mml:math id="M45"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula>, <inline-formula><mml:math id="M46"><mml:mrow><mml:msubsup><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula>, <inline-formula><mml:math id="M47"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B1;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="M48"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003B2;</mml:mo></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:math></inline-formula> using either of the methods in the previous subsections. Given a hypothetical medication input <bold>x</bold><sub>&#x0002A;</sub>, the treatment response for the new subject can then be predicted through the posterior predictive distribution</p>
<disp-formula id="E12"><label>(10)</label><mml:math id="M49"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mo class="qopname">Pr</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mo class="qopname">t</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x003BD;</mml:mo></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x003C4;</mml:mo></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mover accent="true"><mml:mo>&#x003B2;</mml:mo><mml:mo class="qopname">^</mml:mo></mml:mover><mml:mo>*</mml:mo></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mover accent="true"><mml:mo>&#x003B1;</mml:mo><mml:mo class="qopname">^</mml:mo></mml:mover><mml:mo>*</mml:mo></mml:msub></mml:mrow></mml:mfrac><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>I</mml:mtext></mml:mstyle><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mo>&#x0039B;</mml:mo></mml:mrow><mml:mo class="qopname">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x022BA;</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the number of degrees of freedom for the Student&#x00027;s <italic>t</italic>-distribution is given by <inline-formula><mml:math id="M50"><mml:mo>&#x003BD;</mml:mo><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mrow><mml:msub><mml:mover accent="true"><mml:mo>&#x003B1;</mml:mo><mml:mo>^</mml:mo></mml:mover><mml:mo>*</mml:mo></mml:msub></mml:mrow></mml:math></inline-formula>.</p>
<p>Equation (10) may be used to predict the treatment response for this new patient over their course of the treatment directly without learning; that is to treat each appointment as independent and the parameters are not updated. On the other hand, it is possible to perform incremental Bayesian learning over the course of treatment, by using Equation (A5) in <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref> to update the parameters given the treatment outcome measured for each new visit and the associated inputs. Through incremental learning, the model corrects for discrepancies between the true profile and the virtual patient profile of the new patient. As such, one would expect the prediction to improve as data from more visits to the clinic become available. The implementation of these methods is discussed in more detail in Section 4.4.</p>
<p>At some point, the profile of the new patient in terms of the treatment outcome, inputs, and baseline characteristics can be added to the pool of existing patient profiles (training set) to improve the model&#x00027;s generalizability for future patients.</p>
</sec>
<sec>
<title>3.5. Training and validation</title>
<p>As discussed in Section 3.1, 157 patients with longitudinal data were randomized and 10-fold cross-validation partitions were constructed resulting in 10-folds of training-validation data partitions.</p>
<p>First, for each fold, the framework detailed in Section 3.2 was followed and Bayesian linear regression was performed to fit patient-specific parameters to each patient in the training dataset. Second, either of the methods specified in Section 3.3 was used in order to construct virtual patient profiles for each patient in the validation set, using the patient-specific parameters. Finally, the procedure outlined in Section 3.4 was followed in order to obtain a prediction for patients in the validation set; effectively treating each patient as new.</p>
</sec>
<sec>
<title>3.6. Dichotomous remission prediction</title>
<p>Although the model was initially formulated to predict a continuous scale of symptom scores, one can explore dichotomizing the outcome into patients who have shown reduced symptoms and those who have not. The justification is that clinicians and doctors are less likely to be interested in a predicted SNAP-IV score or symptom severity scale as opposed to a simple &#x0201C;yes/no&#x0201D; answer as to whether the patient will be in remission for a given medication. A simple way to adapt the current model to do this is to apply a threshold to the continuous symptom score prediction, below which the patient is predicted to be in remission.</p>
<p>Some of the literature loosely defines remission in ADHD as having a large majority of SNAP-IV responses rated in category 0 (not at all) or 1 (a little) (Hechtman, <xref ref-type="bibr" rid="B32">2005</xref>; Chou et al., <xref ref-type="bibr" rid="B21">2012</xref>). Therefore in this paper, the thresholds were chosen such that the approximate continuous symptom score corresponds to the raw responses from the 18-item SNAP-IV questionnaire all lying in category 1. The resulting symptom score thresholds are only slightly different for inattentiveness and hyperactivity (&#x02212;0.97 vs. &#x02212;0.92, see Section 2.3).</p>
<p>Using these thresholds, it was found that the proportion of visits when measurements were taken indicates that remission was relatively rare, 160 out of a total of 1, 147 (13.95%) for the inattentiveness score and 139 (12.1%) for the hyperactivity score. This is expected, as forced titration initially starts with a low medication dosage and one would not expect an effective reduction in symptom ratings to remission levels before the dosage was ramped up in later appointments; in addition, because of medical persistence issues patients can drop out before the clinicians are able to find an effective dose.</p>
<p>Note that all the methods in this paper were first used to predict the continuous symptom scores by regressing the baseline variables, medication prescribed to the treatment response at following appointments. Dichotomized remission prediction only occurs at a later stage. Right censoring (where patients prematurely drop out of dose optimization stage without achieving remission) is therefore not an issue; repression methods can utilize the remaining appointment information to model treatment response regardless of whether remission was achieved or not.</p>
</sec>
</sec>
<sec id="s4">
<title>4. Performance metrics</title>
<p>To facilitate a comparison between the performance of the different approaches, several performance metrics were used. For the regression tasks, one is interested in the deviation in the predicted symptom scores against the true symptom scores; whereas for the remission classification tasks, one is interested in the performance of the classifiers with regard to the probabilities or ratios of true positive, false positive, true negative and false negative cases.</p>
<sec>
<title>4.1. Regression task</title>
<p>The root mean squared (rms) error measure is defined as the square of the averaged squared error across the 10-folds, across subjects and across all appointments for each individual, i.e.,</p>
<disp-formula id="E13"><label>(11)</label><mml:math id="M51"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msqrt><mml:mrow><mml:mtext class="textrm" mathvariant="normal">rms</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>10</mml:mn><mml:mo>|</mml:mo><mml:mo>&#x1D54A;</mml:mo><mml:mo>|</mml:mo></mml:mrow></mml:mfrac><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x02200;</mml:mo><mml:mi>s</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mo>&#x1D54A;</mml:mo></mml:mrow></mml:munder></mml:mstyle><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:msubsup><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mo>|</mml:mo><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mo>&#x00177;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:msqrt></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x1D54A; is the set of all subjects considered (e.g., those in the validation set), |&#x1D54A;| denotes the number of subjects in &#x1D54A;; <italic>y</italic><sub><italic>s</italic></sub>, &#x00177;<sub><italic>s</italic></sub>, and <italic>A</italic><sub><italic>s</italic></sub> are, respectively, the true outcome symptom score, the fitted or predicted outcome symptom score, and the total number of appointments for the individual subject <italic>s</italic>.</p>
</sec>
<sec>
<title>4.2. Classification task</title>
<sec>
<title>4.2.1. Sensitivity and specificity</title>
<p>The sensitivity (SEN, also known as the true positive rate or recall) is defined as <italic>N</italic><sub>TP</sub>/<italic>N</italic><sub>P</sub> where <italic>N</italic><sub>TP</sub> is the number of <italic>true positives</italic>&#x02014;appointments where measurements indicated remission and were correctly predicted as such; and <italic>N</italic><sub>P</sub> is the actual number of positive cases, i.e., the number of appointments where the corresponding subjects were indeed in remission. This is reported in addition to the specificity (SPC, also known as the true negative rate or fall-out), defined as <italic>N</italic><sub>TN</sub>/<italic>N</italic><sub>N</sub>, where <italic>N</italic><sub><italic>TN</italic></sub> is the number of <italic>true negatives</italic>&#x02014;those <italic>not</italic> in remission and correctly predicted as such; and <italic>N</italic><sub><italic>N</italic></sub> is the actual number of negative cases (Fletcher and Fletcher, <xref ref-type="bibr" rid="B25">2005</xref>). Note that if one lets <italic>N</italic><sub>FP</sub> and <italic>N</italic><sub>FN</sub> be the number of false positives and false negatives respectively, then <italic>N</italic><sub>P</sub> &#x0003D; <italic>N</italic><sub>TP</sub> &#x0002B; <italic>N</italic><sub>FN</sub> and <italic>N</italic><sub>N</sub> &#x0003D; <italic>N</italic><sub>TN</sub> &#x0002B; <italic>N</italic><sub>FP</sub> (Fletcher and Fletcher, <xref ref-type="bibr" rid="B25">2005</xref>).</p>
<p>Sensitivity characterizes the ability of a classifier to rule out false negative predictions (type-II errors) given that a condition is true. On the other hand, specificity measures the ability of a classifier to rule out false positive predictions (type-I errors) given that a condition is false. In this exercise, the sensitivity measure is more important; due to the rarity of remission, and the goal is to try to predict what level of medication is required to achieve remission, the ability of a classifier to recall remission cases (ruling out type-II errors) is more important than ruling out type-I errors.</p>
</sec>
<sec>
<title>4.2.2. PPV and NPV</title>
<p>The positive predictive value (PPV, also known as the precision) is the proportion of true positives in the <italic>predicted</italic> positive cases and is the probability of remission given a positive prediction by the algorithm. As such, the PPV is a measure of the &#x0201C;quality&#x0201D; of a given positive prediction. PPV is given by <italic>N</italic><sub>TP</sub>/(<italic>N</italic><sub>TP</sub> &#x0002B; <italic>N</italic>FP). Conversely, the negative predictive value (NPV) is the proportion of true negatives in the <italic>predicted</italic> negative cases, and is the probability of non-remission given a negative prediction. NPV is given by <italic>N</italic><sub>TN</sub>/(<italic>N</italic><sub>TN</sub> &#x0002B; <italic>N</italic>FN) (Fletcher and Fletcher, <xref ref-type="bibr" rid="B25">2005</xref>).</p>
<p>By the argument outlined above, the PPV is more important for this exercise than the NPV.</p>
</sec>
<sec>
<title>4.2.3. Balanced accuracy</title>
<p>The overall accuracy of a dichotomous predictor is defined by</p>
<disp-formula id="E14"><mml:math id="M52"><mml:mtext class="textrm" mathvariant="normal">Accuracy</mml:mtext><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mtext class="textrm" mathvariant="normal">TP</mml:mtext></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mtext class="textrm" mathvariant="normal">TN</mml:mtext></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>/</mml:mo><mml:mi>N</mml:mi><mml:mo>,</mml:mo></mml:math></disp-formula>
<p>where <italic>N</italic> &#x0003D; 1, 147 is the total number of appointments across all subjects.</p>
<p>However, the overall accuracy measure is known to be problematic when the prevalence of success/failure is low (Alberg et al., <xref ref-type="bibr" rid="B2">2004</xref>), i.e., the data are imbalanced (see also the end of Section 3.6). Due to this, some of the literature uses the balance accuracy (BAC) measure, defined as the average of sensitivity and specificity (Brodersen et al., <xref ref-type="bibr" rid="B15">2010</xref>). This is the accuracy measure used throughout this paper. Note that, numerically, the BAC is closely related to the Youden&#x00027;s <italic>J</italic>-statistic (Youden, <xref ref-type="bibr" rid="B68">1950</xref>), also known as &#x0201C;informedness&#x0201D; or &#x0201C;DeltaP&#x02032;&#x0201D; (Powers, <xref ref-type="bibr" rid="B45">2011</xref>), since it is equal to sensitivity plus specificity minus one.</p>
</sec>
<sec>
<title>4.2.4. ROC and AUC</title>
<p>The receiver operating characteristic (ROC) curve is commonly used in the medical and the machine learning community to evaluate the performance of binary classifiers Fawcett (<xref ref-type="bibr" rid="B24">2006</xref>). It plots the true positive rate (sensitivity) against the false positive rate (one minus specificity) for a given classifier. A curve is obtained when its classification performance can be tuned through setting a threshold or changing a parameter, trading off the true positive rate against the false positive rate. Binary classifiers that can achieve good compromise between sensitivity and specificity have a large area-under-the-curve (AUC), and this single metric may be used to compare the performance between the different classifiers Bradley (<xref ref-type="bibr" rid="B14">1997</xref>).</p>
</sec>
</sec>
<sec>
<title>4.3. Trading off sensitivity and specificity</title>
<p>From Section 3.6, the proportion of appointments without remission is (100&#x02212;13.95)% &#x0003D; 86.05%. Therefore, given this statistic, one would expect that a null model guessing the result randomly would have a sensitivity of 13.95% and a specificity of 86.05%. Simply using point estimates of the continuous symptom score from the learning in the model space approach and thresholding them to give dichotomous predictions of remission results in classifiers with low sensitivity values between 22 and 28% and high specificity values of 94&#x02013;97%. Due to the low number of remission cases compared to the non-remission cases, the classification is biased against predicting the remission cases, leading to low sensitivity (but high specificity). A classifier can be tuned to improve its sensitivity performance by trading off specificity to a certain degree. A good compromise would be maximizing both sensitivity and specificity equally, which is in essence maximizing the BAC or the Youden&#x00027;s <italic>J</italic>-statistic in Section 4.2.3.</p>
<p>The thresholds for remission are defined by the SNAP-IV symptom factor scores as in Section 2.3, and this defines the ground truth of whether a patient is in remission or not. However, one can take advantage of the fact that the predicted continuous symptom scores from the learning in the model space approach form full posterior distributions with uncertainties associated, and the levels of uncertainty are known (e.g., see the error bars in Figures <xref ref-type="fig" rid="F5">5</xref>, <xref ref-type="fig" rid="F6">6</xref>). One may define a critical value as the lower bound of the prediction, above which the probability of the prediction being correct is <italic>x%</italic>. Instead of the remission thresholds comparing against the point estimates, they may be compared against the point estimates minus a critical value. The larger the critical value, the higher the prediction score has to be in order to be classified as not in remission. This in effect is equivalent to raising the threshold, classifying more and more cases into remission, which increases the sensitivity and lowers the specificity. The range of &#x0201C;thresholds&#x0201D; or classifier parameter settings that makes this trade-off can be used to generate a ROC plot (Section 4.2.4). A similar trade-off can be made with classical machine learning algorithms and will be discussed in Section5.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p><bold>Examples of Bayesian linear regression on continuous inattentiveness (INA) and hyperactivity (HYP) symptom scores with the Bayesian linear regression training dataset (<monospace><bold>BRR</bold></monospace>)</bold>. <bold>(A)</bold> Subject &#x00023;11, <bold>(B)</bold> Subject &#x00023;74.</p></caption>
<graphic xlink:href="fphys-08-00199-g0005.tif"/>
</fig>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p><bold>Examples of continuous hyperactivity symptom score prediction with the validation set</bold>. <bold>(A)</bold> Subject &#x00023;6. <bold>(B)</bold> Subject &#x00023;148.</p></caption>
<graphic xlink:href="fphys-08-00199-g0006.tif"/>
</fig>
<p>The training data are used to find an optimal classifier setting in order to achieve the highest BAC, and the same classifier setting is then used to classify the validation data. This ensures that the validation data are not used to minimize the validation error.</p>
</sec>
<sec>
<title>4.4. Benchmarking and implementation</title>
<p>The Bayesian learning in the model space approach relies on prior knowledge (Section 3.2.1) and virtual patient profiles in the model space (Section 3.3), as well as iterative learning (Bayesian update) in order to function. To assess whether these components contribute to the prediction capability of the model, several implementation strategies are investigated, namely:</p>
<list list-type="order">
<list-item><p><bold>Appointment-independent prediction (<monospace>AI</monospace>):</bold> Treating each appointment as independent (as the first appointment) and giving a prediction only using the virtual patient profile;</p></list-item>
<list-item><p><bold>Incremental Bayesian linear regression (<monospace>BR</monospace>):</bold> The first prediction is performed in exactly the same manner as the appointment-independent case. Then, Bayesian linear regression using elicited priors (see Section 3.2.1) is applied progressively. That is, the effect/outcome of medication prescribed in appointment 1, then observed at appointment 2, is used in the regression model. Then, at appointment 3, outcomes from appointments 1 and 2 are used. Similarly, at appointment 4, outcomes from appointments 1, 2 and 3 are used, and so on. This means that except for the first prediction, the incremental Bayesian linear regression learns from scratch the patient-specific parameters (at each appointment) using the elicited priors. This essentially disregards any information already learnt from the current training set (the virtual patient profiles), treats the validation set as a new &#x0201C;training&#x0201D; set, and performs basic Bayesian linear regression fitting. However, instead of all appointment outcomes being available for each new patient, as is the case during the training phase, one simulates the fact that information is progressively collected during the course of treatment for new patients. Since the virtual patient profiles are not utilized, this serves as a benchmark reference to evaluate the effectiveness of the constructed virtual patient profiles when compared with the next case; this represents a method that can be implemented even when no training data exist.</p></list-item>
<list-item><p><bold>Incremental Bayesian learning/update (<monospace>BU</monospace>):</bold> The first appointment is predicted as for the previous two cases, but then when the true value is observed (in appointment 2), it is fed-back into the Bayesian learning model, i.e., Equation (A5) in <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>. This updated model is then used to generate a prediction. This progressive updating continues up to the most recent appointment. The crucial difference between this method and the <monospace>BR</monospace> method is that, here, the priors used were derived from the virtual patient profiles, as opposed to the elicited priors used in the <monospace>BR</monospace> method. When compared to the <monospace>BR</monospace> case, this highlights whether the model space offers any utility in aiding the prediction of treatment response.</p></list-item>
</list>
<p>The Bayesian approach was implemented <italic>ad hoc</italic> in MATLAB software with custom routines. The performance of the prediction was compared across the three different implementations above, using the validation data.</p>
</sec>
</sec>
<sec id="s5">
<title>5. Comparison with conventional methods</title>
<p>To provide context to the results achieved using the learning in the model space approach, the performance of conventional linear regression methods and machine learning methods was also investigated.</p>
<sec>
<title>5.1. Mixed effects models</title>
<p>Linear mixed effects models (MEM) are widely used in many fields; for instance, biology (Rico et al., <xref ref-type="bibr" rid="B49">2007</xref>), ecology (Stevens et al., <xref ref-type="bibr" rid="B58">2007</xref>), linguistics (Nooteboom and Quen&#x000E9;, <xref ref-type="bibr" rid="B42">2008</xref>) and social sciences (Kliegl et al., <xref ref-type="bibr" rid="B37">2009</xref>). They extend upon classical linear regression techniques to support data that have some form of grouping. For example, in this paper, each patient had one or more clinical appointments, and the data from each subject form a group. For each patient <italic>s</italic> with a number of appointments (from 1 to <italic>A</italic><sub><italic>s</italic></sub>), the severity symptom score vector <bold>y</bold><sub><italic>s</italic></sub> (as in Equation 2) is, for simplicity, assumed to have a linear relationship with the baseline and treatment effect via the following formulation:</p>
<disp-formula id="E15"><label>(12)</label><mml:math id="M53"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>q</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>&#x003C9;</mml:mo></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mo>&#x003F5;</mml:mo><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <inline-formula><mml:math id="M54"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>X</mml:mtext></mml:mstyle></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is similar in structure to <bold>X</bold> defined in Equation (4) but without the last column of ones; &#x003C9;<sub><italic>s</italic></sub> &#x02208; &#x0211D;<sub><italic>P</italic>&#x000D7;1</sub> is the subject-specific parameter vector for the fixed effects, <inline-formula><mml:math id="M55"><mml:msub><mml:mrow><mml:mi>q</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0007E;</mml:mo><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x003C3;</mml:mo></mml:mrow><mml:mrow><mml:mi>q</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> is the random effect affecting only the intercept <italic>b</italic><sub>0</sub>, and <inline-formula><mml:math id="M56"><mml:mo>&#x003F5;</mml:mo><mml:mo>&#x0007E;</mml:mo><mml:mi>N</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x003C3;</mml:mo></mml:mrow><mml:mrow><mml:mo>&#x003F5;</mml:mo><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mo>&#x0211D;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x000D7;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is an error term assumed to have a normal distribution. Observe that in this model, the scalar intercept <italic>b</italic><sub>0</sub> is a fixed component for all of the patients in the population and the random effect <italic>a</italic><sub><italic>s</italic></sub> is a subject-specific scalar and is not grouped under any other parameters.</p>
<p>A linear mixed effects model was constructed within the R software (R Core Team, <xref ref-type="bibr" rid="B46">2013</xref>), using the package &#x0201C;<italic>lme4</italic>&#x0201D; (Bates et al., <xref ref-type="bibr" rid="B12">2015</xref>). Severity score predictions were produced by performing out-of-sample forecasts, i.e., on the validation data for each of the folds, using the &#x0201C;<italic>predict</italic>&#x0201D; function in the R software. To generate a prediction, the random effects are assumed to be zero and the population intercept was used. The continuous symptom score regression results for the MEM are prefixed <monospace>MER</monospace>.</p>
<p>For a dichotomized clinical remission classification, the symptom score thresholds 0.92 and 0.97 from Section 2.3 for hyperactivity and inattentiveness were used to generate the ground truths. Following the rationale in Section 4.3, the symptom scores predicted by the MEM are given thresholds at different levels to produce a set of classifiers trading off sensitivity against specificity. These threshold-adjusted classifiers are labeled <monospace>taMEC</monospace>. The best (in terms of Youden&#x00027;s statistic) threshold settings found using the training data were used for the validation data; the thresholds were 0.05 and -0.20, respectively, for the inattentiveness and hyperactivity symptom scores. In addition, the &#x0201C;<italic>melogit</italic>&#x0201D; function in the Stata software (StataCorp, <xref ref-type="bibr" rid="B57">2015</xref>) was used to directly estimate a mixed effects logistic regression model&#x02014;a MEM with a logistic link function that predicts the probability of the binary remission outcome. In this case, the threshold procedure was applied to the probabilities rather than the raw symptom scores. The resulting classifier is labeled <monospace>lrMEC</monospace>.</p>
</sec>
<sec>
<title>5.2. Support vector machines and Gaussian processes</title>
<p>In addition to MEM, machine learning classification approaches using support vector machines (SVM) and Gaussian processes (GP) were benchmarked. Both the SVM and GP learning methods are kernel machines and were implemented using linear (dot product kernel: <italic>k</italic>(<bold>x</bold><sub><italic>i</italic></sub>, <bold>x</bold><sub><italic>j</italic></sub>) &#x0003D; (<bold>x</bold><sub><italic>i</italic></sub>.<bold>x</bold><sub><italic>j</italic></sub>)) and nonlinear kernels (the Gaussian kernel: <inline-formula><mml:math id="M57"><mml:mstyle class="text"><mml:mtext class="textsl" mathvariant="italic">k</mml:mtext></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mo>&#x003B3;</mml:mo><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mo>&#x02225;</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x02225;</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math></inline-formula>). Readers are invited to refer to Burges (<xref ref-type="bibr" rid="B17">1998</xref>) for a detailed description of support vector machines and to Rasmussen and Nickisch (<xref ref-type="bibr" rid="B47">2010</xref>) for a detailed description of GP learning. Compared to MEM and the learning in the model space approach, the SVM and GP are non-parametric methods&#x02014;there are no subject specific parameters to identify; the models map the subject-specific inputs, such as baseline characteristics and the medication dosage, to the output symptom scores.</p>
<p>For the SVM, nested cross-validation was employed to optimize the parameters in the model. A broad log range spanning [10<sup>&#x02212;3</sup>:10<sup>2</sup>] was arbitrarily chosen as the search range for the regularization parameter <italic>C</italic>. Similarly, the gamma parameter of the Gaussian kernel was optimized in the log range spanning [10<sup>&#x02212;4</sup>:10<sup>1</sup>]. For the GP, the model parameters were optimized using conjugate gradient descent, avoiding the need for nested cross-validation. SVM and GP learning approaches were employed as regression models (support vector regression <monospace>SVR</monospace> and Gaussian process regression <monospace>GPR</monospace>) for the linear and nonlinear kernels to predict the clinical scores. Dichotomous remission predictions were obtained by thresholding the distance from the hyperplane for the SVM, and for thresholding the probabilistic predictions of class membership for GP. These binary classifiers are respectively labeled as <monospace>SVC</monospace> and <monospace>GPC</monospace>.</p>
<p>From Section 3.6, the number of remission cases outweighed non-remission cases by a ratio of roughly 1:7. For a classification task, this imbalance of data is problematic for many classification algorithms (He and Garcia, <xref ref-type="bibr" rid="B31">2009</xref>). To help alleviate this, a downsampling approach was implemented for both linear and nonlinear kernels of the SVM and GP classifiers <monospace>dsSVC</monospace> and <monospace>dsGPC</monospace>. During the training phase, the non-responder class was downsampled randomly to match the number of training instances in the remission class. By repeating this downsampling procedure, an ensemble of 1,000 classifiers was trained. A classification prediction was generated by majority voting of the ensemble. Additionally, for the SVMs, an alternative is to learn the regularization parameters <italic>C</italic> on a per-class basis. The rationale is that a higher penalty for errors can be placed on the more abundant class (Osuna et al., <xref ref-type="bibr" rid="B44">1997</xref>); this method is referred to as the weighted SVM (<monospace>rwSVC</monospace>). The per-class <italic>C</italic> parameters <italic>C</italic><sup>&#x0002B;</sup> and <italic>C</italic><sup>&#x02212;</sup> were optimized using the ranges [10<sup>&#x02212;3</sup>:10<sup>1</sup>] and [10<sup>&#x02212;2</sup>:10<sup>2</sup>], respectively. Finally, for the Gaussian process classifier, it is possible to calibrate the probabilistic predictions in order to help account for imbalanced data (Bishop, <xref ref-type="bibr" rid="B13">2006</xref>); this approach is referred to as a re-calibrated GP (<monospace>rcGPC</monospace>).</p>
<p>Using <sc>Matlab</sc> software, the SVM was implemented using the <monospace>libsvm</monospace> toolbox (Chang and Lin, <xref ref-type="bibr" rid="B19">2011</xref>), the Gaussian process learning was implemented using the <monospace>GPML</monospace> toolbox (Rasmussen and Nickisch, <xref ref-type="bibr" rid="B47">2010</xref>).</p>
</sec>
</sec>
<sec id="s6">
<title>6. Results and discussions</title>
<sec>
<title>6.1. Continuous symptom score prediction</title>
<p>The rms errors across all models are reported in Table <xref ref-type="table" rid="T1">1</xref>.</p>
<table-wrap-group position="float" id="T1">
<label>Table 1</label>
<caption><p><bold>Rms errors for predicting symptom scores for inattentiveness (INA) and hyperactivity (HYP) using the (A) learning in model space and (B) conventional approaches</bold>.</p></caption>
<table-wrap>
<table frame="hsides" rules="groups">
<thead>
<tr style="border-bottom: thin solid #000000;">
<th valign="top" align="left" colspan="5"><bold>(1A) Learning in model space</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><bold>Inattentiveness</bold></th>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><bold>Hyperactivity</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td/>
<td valign="top" align="center"><bold>Method 1</bold></td>
<td valign="top" align="center"><bold>Method 2</bold></td>
<td valign="top" align="center"><bold>Method 1</bold></td>
<td valign="top" align="center"><bold>Method 2</bold></td>
</tr>
<tr style="border-top: thin solid #000000;">
<td valign="top" align="left"><monospace>AIR</monospace><xref ref-type="table-fn" rid="TN1"><sup>&#x0002A;</sup></xref></td>
<td valign="top" align="center">0.98</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.85</td>
</tr>
<tr>
<td valign="top" align="left"><monospace>BRR</monospace><xref ref-type="table-fn" rid="TN2"><sup>&#x02020;</sup></xref></td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.84</td>
</tr>
<tr>
<td valign="top" align="left"><monospace>BUR</monospace><xref ref-type="table-fn" rid="TN3"><sup>&#x02021;</sup></xref></td>
<td valign="top" align="center">0.99</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">1.01</td>
<td valign="top" align="center">0.75</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="TN1">
<label>&#x0002A;</label>
<p><monospace>AIR</monospace><italic>: appointment-independent Bayesian linear prediction</italic>.</p></fn>
<fn id="TN2">
<label>&#x02020;</label>
<p><monospace>BRR</monospace><italic>: retrospective Bayesian linear regression</italic>.</p></fn>
<fn id="TN3">
<label>&#x02021;</label>
<p><monospace>BUR</monospace><italic>: incremental Bayesian learning/update linear regression</italic>.</p></fn>
</table-wrap-foot>
</table-wrap>
<table-wrap>
<table frame="hsides" rules="groups">
<thead>
<tr style="border-bottom: thin solid #000000;">
<th valign="top" align="left" colspan="5"><bold>(1B) Conventional approaches</bold></th>
</tr>
<tr style="border-top: thin solid #000000;">
<th valign="top" align="left"><bold>Kernel</bold></th>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><bold>Inattentiveness</bold></th>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><bold>Hyperactivity</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>Linear</bold></th>
<th valign="top" align="center"><bold>Nonlinear</bold></th>
<th valign="top" align="center"><bold>Linear</bold></th>
<th valign="top" align="center"><bold>Non-linear</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><monospace>SVR</monospace><xref ref-type="table-fn" rid="TN4"><sup>&#x0002A;</sup></xref></td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="left"><monospace>GPR</monospace><xref ref-type="table-fn" rid="TN5"><sup>&#x02020;</sup></xref></td>
<td valign="top" align="center">0.72</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.84</td>
</tr>
<tr>
<td valign="top" align="left"><monospace>MER</monospace><xref ref-type="table-fn" rid="TN6"><sup>&#x02021;</sup></xref></td>
<td valign="top" align="center" colspan="2">0.82</td>
<td valign="top" align="center" colspan="2">0.83</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="TN4">
<label>&#x0002A;</label>
<p><monospace>SVR</monospace><italic>: support vector machine regression</italic>.</p></fn>
<fn id="TN5">
<label>&#x02020;</label>
<p><monospace>GPR</monospace><italic>: Gaussian processes regression</italic>.</p></fn>
<fn id="TN6">
<label>&#x02021;</label>
<p><monospace>MER</monospace><italic>: mixed effects regression</italic>.</p></fn>
</table-wrap-foot>
</table-wrap>
</table-wrap-group>
<sec>
<title>6.1.1. Learning in the model space approach</title>
<p>Looking at the learning in the model space approach, one can observe that the virtual patient profile construction method, labeled Method 2, resulted in lower errors overall compared to Method 1. In Method 1, the mappings were learnt using simple linear regression from baseline variables to the parameter space. In addition to this simple linear regression, low degree polynomial (quadratic to quartic) basis functions were tried; whilst degrees up to a cubic resulted in a slightly lower training error, there was worse generalizability (i.e., higher validation error). For Method 2, the incremental Bayesian learning (<monospace>BUR</monospace>) approach performed the best overall; its performance advantage over the appointment-independent prediction (<monospace>AIR</monospace>) approach is expected given that it allows the model to adapt to a new patient as the treatment continues. The performance advantage over the retrospective Bayesian linear regression (<monospace>BRR</monospace>) approach can be attributed to the fact that the virtual patient profile (Section 3.3) had utilized the prior whilst the Bayesian linear regression only uses the elicited prior (see Section 3.2.1). This supports the fact that the training population was able to add valuable information to the prediction task.</p>
<p>We recall that the <monospace>BUR</monospace> constructs virtual patient profiles while the <monospace>BRR</monospace> only uses the prior knowledge. It is interesting to note that the <monospace>BRR</monospace> outperforms the <monospace>BUR</monospace> using Method 1, suggesting that Method 1 was not an effective method for incorporating information from existing patient models.</p>
<p>Figure <xref ref-type="fig" rid="F7">7</xref> shows the rms values averaged across all subjects during the validation phase and sorted by the clinical appointment (visit) number. Data above 15 visits are not shown as only a single patient had more than 15 visits. There is a slight downward trend visible with the <monospace>BUR</monospace>; suggesting that incremental Bayesian learning approach is able to reduce the prediction error as more data are known about a new patient through repeated appointments. The <monospace>BRR</monospace> also shows a downward trend, but the error is slightly higher than the <monospace>BUR</monospace>. This is because the <monospace>BRR</monospace> starts with only the elicited prior and performs learning (fitting) when more data are available, unlike the <monospace>BUR</monospace> which starts off with information from the training set in the form of a constructed/estimated virtual patient profile.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p><bold>Rms prediction (validation) error averaged across all subjects vs. appointment number</bold>.</p></caption>
<graphic xlink:href="fphys-08-00199-g0007.tif"/>
</fig>
<p>Figure <xref ref-type="fig" rid="F5">5</xref> illustrates some examples of Bayesian linear regression performed during the training phase and their associated fitting rms errors. It can be seen that Bayesian linear regression fits similarly well for both the inattentiveness and hyperactivity symptom scores. Switching to a new type of medicine is usually associated with larger uncertainty (error bars). Looking at subject &#x00023;74 (Figure <xref ref-type="fig" rid="F5">5B</xref>) in particular, it can be seen that, despite having the same input dosage from appointments 6&#x02013;8, there were variations in the severity of the ADHD symptoms. It is not possible to know the exact reason for the variation for this subject during this particular period, without further information&#x02014;perhaps this was due to adherence issues (the patient not taking their medication as prescribed), physiological factors, measurement &#x0201C;noise,&#x0201D; or perhaps something else entirely. By design, Bayesian linear regression can only fit the same outcome given the same input. This does highlight the fact that the current model may not have enough information in the form of covariates to account for some of these factors.</p>
<p>A subset of results for the validation phase is plotted in Figure <xref ref-type="fig" rid="F6">6</xref>. For brevity, only prediction outcomes for hyperactivity symptom scores are shown and retrospective Bayesian linear regression results (<monospace>BRR</monospace>) are omitted. The lack of solid lines connecting the predictions in the topmost subplot serves as a reminder that the model does not incorporate temporal aspects for the case of appointment-independent (<monospace>AIR</monospace>) prediction, which treats each appointment as the first (new) appointment for a new patient. This is also why the 95% confidence intervals for <monospace>AIR</monospace> are larger (more uncertain) than those for the <monospace>BUR</monospace>. Also note that, by design, the prediction results for the first appointment are identical for both approaches.</p>
<p>The figures illustrate that, during prognosis, incremental learning does not always improve the prediction error compared to simply predicting at every appointment without updating the model using new information. However, based on the rms errors in Table <xref ref-type="table" rid="T1">1A</xref>, one expects incremental learning to perform better overall across subjects, especially for subjects with a prediction offset, such as over- or under-estimates. This is illustrated by the results for subject &#x00023;148 given in Figure <xref ref-type="fig" rid="F6">6B</xref>, where the virtual patient profile for this patient consistently underestimates the actual hyperactivity score. Here, the incremental Bayesian learning was able to adapt the parameter &#x003C9; and shifted the prediction upwards, resulting in lower prediction errors over the subsequent appointments.</p>
</sec>
<sec>
<title>6.1.2. Conventional machine learning approaches</title>
<p>Looking at the results in Table <xref ref-type="table" rid="T1">1B</xref>, the conventional approaches yield similar performance, with linear <monospace>SVR</monospace> and <monospace>GPR</monospace> methods performing better than their nonlinear counterparts. The mixed effects model has slightly worse results. Errors of linear <monospace>SCR</monospace> and linear <monospace>GPR</monospace> are similar to each other, and to those for the learning in the model space approach <monospace>BUR</monospace> with Method 2. We conclude that for the task of predicting continuous symptom scores with the dataset investigated, the learning in the model space approach performs comparably with conventional approaches.</p>
</sec>
</sec>
<sec>
<title>6.2. Dichotomous remission prediction</title>
<sec>
<title>6.2.1. Learning in the model space approach</title>
<p>For the learning in the model space approach, the ROC curves for the dichotomous predictor are plotted in Figure <xref ref-type="fig" rid="F8">8</xref> for both of the virtual patient profile (Section 3.3) construction methods. As the results for inattentiveness and hyperactivity scores were similar, only the ROC curves for inattentiveness are shown. The AUC values are given in the legend. The crosses on the lines mark the resulting classifier performance if one uses point estimates for the continuous symptom score from the model and simply applies the clinical remission thresholds. The squares mark the classifiers that have critical values based on maximizing the Youden&#x00027;s <italic>J</italic>-statistic (or the BAC, see Section 4.2.3) for the training set&#x02014;this is equivalent to the sensitivity and specificity measures being maximized equally as a function of the critical values. Lastly, the circles mark the best classifier for the validation set in terms of the Youden&#x00027;s <italic>J</italic>-statistic. The closer the squares are to the circles, the better optimized the classifier is assuming no knowledge of the validation dataset. Those optimized classifiers marked by squares in the graph are used to generate various binary classifier performance metrics (see Section 4.2) in Table <xref ref-type="table" rid="T2">2</xref>.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p><bold>Receiver operator characteristic (ROC) plots of inattentiveness prediction using virtual patient profile constructed by Methods 1 and 2 (A,B)</bold> in the learning in model space approach; crosses: no critical value adjustment (based on point estimates); squares: best performing critical values on training set; circles: best performing critical values on the validation set; AUC: Area under the ROC curve <monospace>AIC</monospace>: appointment-independent classifier; <monospace>BRC</monospace>: retrospective Bayesian linear regression classifier; <monospace>BUC</monospace>: incremental Bayesian learning/update classifier.</p></caption>
<graphic xlink:href="fphys-08-00199-g0008.tif"/>
</fig>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p><bold>Sensitivity, specificity, accuracy, and AUC of the remission classifier with critical values adjusted with respect to uncertainties in the predicted symptom scores</bold>.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th/>
<th valign="top" align="center" colspan="4" style="border-bottom: thin solid #000000;"><bold>Inattentiveness</bold></th>
<th valign="top" align="center" colspan="4" style="border-bottom: thin solid #000000;"><bold>Hyperactivity</bold></th>
</tr>
<tr>
<th/>
<th/>
<th valign="top" align="center" colspan="2"><bold>Method 1</bold></th>
<th valign="top" align="center" colspan="2"><bold>Method 2</bold></th>
<th valign="top" align="center" colspan="2"><bold>Method 1</bold></th>
<th valign="top" align="center" colspan="2"><bold>Method 2</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center" rowspan="6">CFM<xref ref-type="table-fn" rid="TN7"><sup>a</sup></xref></td>
<td valign="top" align="center" rowspan="2"><monospace>AIC</monospace><xref ref-type="table-fn" rid="TN14"><sup>&#x0002A;</sup></xref></td>
<td valign="top" align="center">93</td>
<td valign="top" align="center">327</td>
<td valign="top" align="center">141</td>
<td valign="top" align="center">501</td>
<td valign="top" align="center">82</td>
<td valign="top" align="center">314</td>
<td valign="top" align="center">121</td>
<td valign="top" align="center">448</td>
</tr>
<tr>
<td valign="top" align="center">67</td>
<td valign="top" align="center">660</td>
<td valign="top" align="center">19</td>
<td valign="top" align="center">486</td>
<td valign="top" align="center">57</td>
<td valign="top" align="center">694</td>
<td valign="top" align="center">18</td>
<td valign="top" align="center">560</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="2"><monospace>BRC</monospace><xref ref-type="table-fn" rid="TN15"><sup>&#x02020;</sup></xref></td>
<td valign="top" align="center">140</td>
<td valign="top" align="center">482</td>
<td valign="top" align="center">140</td>
<td valign="top" align="center">483</td>
<td valign="top" align="center">125</td>
<td valign="top" align="center">480</td>
<td valign="top" align="center">125</td>
<td valign="top" align="center">483</td>
</tr>
<tr>
<td valign="top" align="center">20</td>
<td valign="top" align="center">505</td>
<td valign="top" align="center">20</td>
<td valign="top" align="center">504</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">528</td>
<td valign="top" align="center">14</td>
<td valign="top" align="center">525</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="2"><monospace>BUC</monospace><xref ref-type="table-fn" rid="TN16"><sup>&#x02021;</sup></xref></td>
<td valign="top" align="center">117</td>
<td valign="top" align="center">338</td>
<td valign="top" align="center">123</td>
<td valign="top" align="center">280</td>
<td valign="top" align="center">97</td>
<td valign="top" align="center">307</td>
<td valign="top" align="center">109</td>
<td valign="top" align="center">253</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center">43</td>
<td valign="top" align="center">649</td>
<td valign="top" align="center">37</td>
<td valign="top" align="center">707</td>
<td valign="top" align="center">42</td>
<td valign="top" align="center">701</td>
<td valign="top" align="center">30</td>
<td valign="top" align="center">755</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="3">SEN<xref ref-type="table-fn" rid="TN8"><sup>b</sup></xref></td>
<td valign="top" align="center"><monospace>AIC</monospace></td>
<td valign="top" align="center" colspan="2">58.1%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">88.1%</td>
<td valign="top" align="center" colspan="2">59.0%</td>
<td valign="top" align="center" colspan="2">87.1%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>BRC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">87.5%</td>
<td valign="top" align="center" colspan="2">87.5%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">89.9%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">89.9%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>BUC</monospace></td>
<td valign="top" align="center" colspan="2">73.1%</td>
<td valign="top" align="center" colspan="2">76.9%</td>
<td valign="top" align="center" colspan="2">69.8%</td>
<td valign="top" align="center" colspan="2">78.4%</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="3">SPC<xref ref-type="table-fn" rid="TN9"><sup>c</sup></xref></td>
<td valign="top" align="center"><monospace>AIC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">66.9%</td>
<td valign="top" align="center" colspan="2">49.2%</td>
<td valign="top" align="center" colspan="2">68.9%</td>
<td valign="top" align="center" colspan="2">55.6%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>BRC</monospace></td>
<td valign="top" align="center" colspan="2">51.2%</td>
<td valign="top" align="center" colspan="2">51.1%</td>
<td valign="top" align="center" colspan="2">52.4%</td>
<td valign="top" align="center" colspan="2">52.1%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>BUC</monospace></td>
<td valign="top" align="center" colspan="2">65.8%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">71.6%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">69.5%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">74.9%</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="3">BAC<xref ref-type="table-fn" rid="TN10"><sup>d</sup></xref></td>
<td valign="top" align="center"><monospace>AIC</monospace></td>
<td valign="top" align="center" colspan="2">62.5%</td>
<td valign="top" align="center" colspan="2">68.7%</td>
<td valign="top" align="center" colspan="2">63.9%</td>
<td valign="top" align="center" colspan="2">71.3%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>BRC</monospace></td>
<td valign="top" align="center" colspan="2">69.3%</td>
<td valign="top" align="center" colspan="2">69.3%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">71.2%</td>
<td valign="top" align="center" colspan="2">71.0%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>BUC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">69.4%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">74.3%</td>
<td valign="top" align="center" colspan="2">69.7%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">76.7%</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="3">PPV<xref ref-type="table-fn" rid="TN11"><sup>e</sup></xref></td>
<td valign="top" align="center"><monospace>AIC</monospace></td>
<td valign="top" align="center" colspan="2">22.1%</td>
<td valign="top" align="center" colspan="2">22.0%</td>
<td valign="top" align="center" colspan="2">20.7%</td>
<td valign="top" align="center" colspan="2">21.3%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>BRC</monospace></td>
<td valign="top" align="center" colspan="2">22.5%</td>
<td valign="top" align="center" colspan="2">22.5%</td>
<td valign="top" align="center" colspan="2">20.7%</td>
<td valign="top" align="center" colspan="2">20.6%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>BUC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">25.7%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">30.5%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">24.0%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">30.1%</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="3">NPV<xref ref-type="table-fn" rid="TN12"><sup>f</sup></xref></td>
<td valign="top" align="center"><monospace>AIC</monospace></td>
<td valign="top" align="center" colspan="2">90.8%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">96.2%</td>
<td valign="top" align="center" colspan="2">92.4%</td>
<td valign="top" align="center" colspan="2">96.9%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>BRC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">96.2%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">96.2%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">97.4%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">97.4%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>BUC</monospace></td>
<td valign="top" align="center" colspan="2">93.8%</td>
<td valign="top" align="center" colspan="2">95.0%</td>
<td valign="top" align="center" colspan="2">94.4%</td>
<td valign="top" align="center" colspan="2">96.2%</td>
</tr>
<tr>
<td valign="top" align="center" rowspan="3">AUC<xref ref-type="table-fn" rid="TN13"><sup>g</sup></xref></td>
<td valign="top" align="center"><monospace>AIC</monospace></td>
<td valign="top" align="center" colspan="2">69.0%</td>
<td valign="top" align="center" colspan="2">72.0%</td>
<td valign="top" align="center" colspan="2">68.0%</td>
<td valign="top" align="center" colspan="2">73.8%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>BRC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">81.2%</td>
<td valign="top" align="center" colspan="2">80.9%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">83.6%</td>
<td valign="top" align="center" colspan="2">83.3%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>BUC</monospace></td>
<td valign="top" align="center" colspan="2">77.1%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">82.3%</td>
<td valign="top" align="center" colspan="2">76.7%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">84.4%</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="TN7">
<label>a</label>
<p><italic>CFM: confusion matrix</italic>.</p></fn>
<fn id="TN8">
<label>b</label>
<p><italic>SEN: sensitivity</italic>.</p></fn>
<fn id="TN9">
<label>c</label>
<p><italic>SPC: specificity</italic>.</p></fn>
<fn id="TN10">
<label>d</label>
<p><italic>BAC: balanced accuracy</italic>.</p></fn>
<fn id="TN11">
<label>e</label>
<p><italic>PPV: positive predictive value</italic>.</p></fn>
<fn id="TN12">
<label>f</label>
<p><italic>NPV: negative predictive value</italic>.</p></fn>
<fn id="TN13">
<label>g</label>
<p><italic>AUC: area under ROC curve</italic>.</p></fn>
<fn id="TN14">
<label>&#x0002A;</label>
<p><monospace>AIC</monospace><italic>: appointment-independent classifier</italic>.</p></fn>
<fn id="TN15">
<label>&#x02020;</label>
<p><monospace>BRC</monospace><italic>: retrospective Bayesian linear regression classifier</italic>.</p></fn>
<fn id="TN16">
<label>&#x02021;</label>
<p><monospace>BUC</monospace><italic>: incremental Bayesian learning/update classifier</italic>.</p></fn>
<p><italic>Shaded values represent best performance amongst the compared methods</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>Looking at Table <xref ref-type="table" rid="T2">2</xref>. The confusion matrices (CFM) show the number of true positives and false negatives in the first column, and false positives and true negatives in the second column. These may be used to calculate any classification performance metrics not included in this paper, such as the <italic>F</italic>-measure.</p>
<p>Similar to the continuous symptom prediction task, the <monospace>BRC</monospace> outperforms the <monospace>AIC</monospace> showing that posterior information is utilized effectively. As in the continuous task, the virtual patient profile construction method labeled Method 2 is better overall than Method 1, but the difference is much smaller in the classification task and the advantage is not universal across all metrics, especially for the <monospace>AIC</monospace>. Note that the virtual patient profile construction method has little effect on the <monospace>BRC</monospace> as it does not use it. The <monospace>BRC</monospace> achieves higher sensitivity values but a lower PPV compared to the <monospace>BUC</monospace>, meaning that the <monospace>BRC</monospace> is better at recalling remission cases, but the remission predictions by the <monospace>BUC</monospace> are more reliable. The SPC achieved by the <monospace>BUC</monospace> is notably higher, being better at ruling out false positives.</p>
</sec>
<sec>
<title>6.2.2. Conventional machine learning approaches</title>
<p>Table <xref ref-type="table" rid="T3">3</xref> shows the binary classifier performance metrics for the conventional machine learning approaches. Apart from the AUC, all of the other metrics in the table were derived from classifier settings (set-points) that had optimized the balanced accuracy (BAC) during the training stage. Apart from <monospace>rcGPC</monospace>, the BAC values across the different approaches are similar. The <monospace>MEC</monospace> classifiers perform well compared with <monospace>GPC</monospace> and <monospace>SVC</monospace>, with consistently high AUC values for both inattentiveness and hyperactivity. However, the set-points of the <monospace>MEC</monospace> classifiers achieve lower sensitivity (but higher specificity) than the <monospace>SVC</monospace>. As mentioned in Section 4.2, a higher sensitivity is more important for this exercise. PPV is the other measure of interest; the <monospace>lrMEC</monospace>, in particular, achieved the highest PPV amongst all the conventional approaches&#x02014;partially helped by its low sensitivity.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p><bold>Sensitivity, specificity, accuracy, and AUC of the remission classifier with critical values adjusted with respect to uncertainties in the predicted symptom scores</bold>.</p></caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th/>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><bold>Inattentiveness</bold></th>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><bold>Hyperactivity</bold></th>
</tr>
<tr>
<th/>
<th/>
<th valign="top" align="center"><bold>Linear</bold></th>
<th valign="top" align="center"><bold>Non-linear</bold></th>
<th valign="top" align="center"><bold>Linear</bold></th>
<th valign="top" align="center"><bold>Non-linear</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center" rowspan="6">Sensitivity</td>
<td valign="top" align="center"><monospace>dsSVC</monospace><xref ref-type="table-fn" rid="TN17"><sup>a</sup></xref></td>
<td valign="top" align="center">70.0%</td>
<td valign="top" align="center">70.6%</td>
<td valign="top" align="center">67.6%</td>
<td valign="top" align="center">66.9%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>dsGPC</monospace><xref ref-type="table-fn" rid="TN18"><sup>b</sup></xref></td>
<td valign="top" align="center">68.1%</td>
<td valign="top" align="center">67.5%</td>
<td valign="top" align="center">67.6%</td>
<td valign="top" align="center">66.9%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rwSVC</monospace><xref ref-type="table-fn" rid="TN19"><sup>c</sup></xref></td>
<td valign="top" align="center" style="background-color:#e7e7e8">76.9%</td>
<td valign="top" align="center">33.8%</td>
<td valign="top" align="center" style="background-color:#e7e7e8">76.9%</td>
<td valign="top" align="center">69.1%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rcGPC</monospace><xref ref-type="table-fn" rid="TN20"><sup>d</sup></xref></td>
<td valign="top" align="center">43.1%</td>
<td valign="top" align="center">43.2%</td>
<td valign="top" align="center">71.3%</td>
<td valign="top" align="center">48.2%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>taMEC</monospace><xref ref-type="table-fn" rid="TN21"><sup>e</sup></xref></td>
<td valign="top" align="center" colspan="2">60.0%</td>
<td valign="top" align="center" colspan="2">58.9%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>lrMEC</monospace><xref ref-type="table-fn" rid="TN22"><sup>f</sup></xref></td>
<td valign="top" align="center" colspan="2">53.1%</td>
<td valign="top" align="center" colspan="2">56.0%</td>
</tr>
<tr>
<td valign="middle" align="center" rowspan="6">Specificity</td>
<td valign="top" align="center"><monospace>dsSVC</monospace></td>
<td valign="top" align="center">61.9%</td>
<td valign="top" align="center">62.3%</td>
<td valign="top" align="center">67.8%</td>
<td valign="top" align="center">66.6%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>dsGPC</monospace></td>
<td valign="top" align="center">67.9%</td>
<td valign="top" align="center">69.1%</td>
<td valign="top" align="center">67.8%</td>
<td valign="top" align="center">71.3%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rwSVC</monospace></td>
<td valign="top" align="center">55.9%</td>
<td valign="top" align="center">77.6%</td>
<td valign="top" align="center">62.1%</td>
<td valign="top" align="center">62.1%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rcGPC</monospace></td>
<td valign="top" align="center">59.0%</td>
<td valign="top" align="center">18.4%</td>
<td valign="top" align="center">50.7%</td>
<td valign="top" align="center">50.6%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>taMEC</monospace></td>
<td valign="top" align="center" colspan="2">73.3%</td>
<td valign="top" align="center" colspan="2">73.8%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>lrMEC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">81.4%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">81.4%</td>
</tr>
<tr>
<td valign="middle" align="center" rowspan="6">Balanced accuracy</td>
<td valign="top" align="center"><monospace>dsSVC</monospace></td>
<td valign="top" align="center">66.0%</td>
<td valign="top" align="center">66.5%</td>
<td valign="top" align="center">67.7%</td>
<td valign="top" align="center">66.7%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>dsGPC</monospace></td>
<td valign="top" align="center">68.0%</td>
<td valign="top" align="center" style="background-color:#e7e7e8">68.3%</td>
<td valign="top" align="center">67.7%</td>
<td valign="top" align="center">69.1%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rwSVC</monospace></td>
<td valign="top" align="center">66.4%</td>
<td valign="top" align="center">55.7%</td>
<td valign="top" align="center">69.5%</td>
<td valign="top" align="center">65.6%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rcGPC</monospace></td>
<td valign="top" align="center">51.1%</td>
<td valign="top" align="center">44.8%</td>
<td valign="top" align="center">46.9%</td>
<td valign="top" align="center">49.4%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>taMEC</monospace></td>
<td valign="top" align="center" colspan="2">66.6%</td>
<td valign="top" align="center" colspan="2">66.4%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>lrMEC</monospace></td>
<td valign="top" align="center" colspan="2">67.2%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">70.6%</td>
</tr>
<tr>
<td valign="middle" align="center" rowspan="6">Positive predictive value</td>
<td valign="top" align="center"><monospace>dsSVC</monospace></td>
<td valign="top" align="center">23.0%</td>
<td valign="top" align="center">23.3%</td>
<td valign="top" align="center">22.4%</td>
<td valign="top" align="center">21.6%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>dsGPC</monospace></td>
<td valign="top" align="center">25.6%</td>
<td valign="top" align="center">26.2%</td>
<td valign="top" align="center">22.4%</td>
<td valign="top" align="center">24.3%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rwSVC</monospace></td>
<td valign="top" align="center">22.0%</td>
<td valign="top" align="center">19.6%</td>
<td valign="top" align="center">21.9%</td>
<td valign="top" align="center">20.1%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rcGPC</monospace></td>
<td valign="top" align="center">15.6%</td>
<td valign="top" align="center">12.4%</td>
<td valign="top" align="center">10.8%</td>
<td valign="top" align="center">11.9%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>taMEC</monospace></td>
<td valign="top" align="center" colspan="2">26.7%</td>
<td valign="top" align="center" colspan="2">23.7%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>lrMEC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">31.6%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">29.0%</td>
</tr>
<tr>
<td valign="middle" align="center" rowspan="6">Negative predictive value</td>
<td valign="top" align="center"><monospace>dsSVC</monospace></td>
<td valign="top" align="center">92.7%</td>
<td valign="top" align="center">92.9%</td>
<td valign="top" align="center">93.8%</td>
<td valign="top" align="center">93.9%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>dsGPC</monospace></td>
<td valign="top" align="center">92.9%</td>
<td valign="top" align="center">92.9%</td>
<td valign="top" align="center">93.8%</td>
<td valign="top" align="center">93.4%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rwSVC</monospace></td>
<td valign="top" align="center" style="background-color:#e7e7e8">93.7%</td>
<td valign="top" align="center">87.8%</td>
<td valign="top" align="center" style="background-color:#e7e7e8">95.1%</td>
<td valign="top" align="center">93.6%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rcGPC</monospace></td>
<td valign="top" align="center">86.5%</td>
<td valign="top" align="center">79.8%</td>
<td valign="top" align="center">86.6%</td>
<td valign="top" align="center">87.63%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>taMEC</monospace></td>
<td valign="top" align="center" colspan="2">91.8%</td>
<td valign="top" align="center" colspan="2">92.9%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>lrMEC</monospace></td>
<td valign="top" align="center" colspan="2">91.5%</td>
<td valign="top" align="center" colspan="2">94.1%</td>
</tr>
<tr>
<td valign="middle" align="center" rowspan="6">Area under ROC curve</td>
<td valign="top" align="center"><monospace>dsSVC</monospace></td>
<td valign="top" align="center">71%</td>
<td valign="top" align="center">69%</td>
<td valign="top" align="center">73%</td>
<td valign="top" align="center">71%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>dsGPC</monospace></td>
<td valign="top" align="center">75%</td>
<td valign="top" align="center">71%</td>
<td valign="top" align="center">73%</td>
<td valign="top" align="center">70%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rwSVC</monospace></td>
<td valign="top" align="center">71%</td>
<td valign="top" align="center">60%</td>
<td valign="top" align="center">76%</td>
<td valign="top" align="center">71%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>rcGPC</monospace></td>
<td valign="top" align="center">49%</td>
<td valign="top" align="center">41%</td>
<td valign="top" align="center">46%</td>
<td valign="top" align="center">48%</td>
</tr>
<tr>
<td valign="top" align="center"><monospace>taMEC</monospace></td>
<td valign="top" align="center" colspan="2">74.8%</td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">77.5%</td>
</tr>
<tr style="border-bottom: thin solid #000000;">
<td valign="top" align="center"><monospace>lrMEC</monospace></td>
<td valign="top" align="center" colspan="2" style="background-color:#e7e7e8">75.8%</td>
<td valign="top" align="center" colspan="2">77.2%</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="TN17">
<label>a</label>
<p><italic><monospace>dsSVC</monospace>: down-sampled support vector machine classifier;</italic></p></fn>
<fn id="TN18">
<label>b</label>
<p><italic><monospace>dsGPC</monospace>: down-sampled Gaussian processes classifier;</italic></p></fn>
<fn id="TN19">
<label>c</label>
<p><italic><monospace>rwSVC</monospace>: regularization-weighted support vector machine classifier;</italic></p></fn>
<fn id="TN20">
<label>d</label>
<p><italic><monospace>rwGPC</monospace>: regularization-weighted support Gaussian processes classifier;</italic></p></fn>
<fn id="TN21">
<label>e</label>
<p><italic><monospace>taMEC</monospace>: threshold-adjusted mixed effects classifier;</italic></p></fn>
<fn id="TN22">
<label>f</label>
<p><italic><monospace>lrMEC</monospace>: logitic regression mixed effects classifier</italic>.</p></fn>
<p><italic>Shaded values represent best performance amongst the compared methods</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>The ROC plot for the <monospace>MEC</monospace> is shown in Figure <xref ref-type="fig" rid="F9">9</xref>. The <monospace>lrMEC</monospace> variant fitted the training set better but both the <monospace>lrMEC</monospace> and the <monospace>taMEC</monospace> achieve similar validation performance. Tracing the ROC values, the <monospace>lrMEC</monospace> seems more suitable for high specificity settings while <monospace>taMEC</monospace> appears to be more suitable for high sensitivity classification.</p>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p><bold>Receiver operator characteristic (ROC) plot of inattentiveness prediction using mixed effects models; crosses: no threshold adjustment (based on point estimates); squares: best performing threshold setting on training set; circles: best performing threshold setting on the validation set; AUC: Area under the ROC curve; <monospace><bold>taMEC</bold></monospace>: threshold-adjusted mixed effects classifier; <monospace><bold>lrMEC</bold></monospace>: logitic regression mixed effects classifier</bold>.</p></caption>
<graphic xlink:href="fphys-08-00199-g0009.tif"/>
</fig>
<p>For the machine learning approaches <monospace>GPC</monospace> and <monospace>SVC</monospace>, linear models work better. This was similarly observed in the continuous symptom score prediction task. Comparing methods in tackling data imbalance, the weighted SVM classifier <monospace>rwSVC</monospace> method performed better than the downsampled <monospace>dsSVC</monospace> method, while the downsampled Gaussian process classifier <monospace>dsGPC</monospace> method performed better than the re-calibrated <monospace>rcGPC</monospace>. Looking at both the BAC and AUC metrics, <monospace>rwSVC</monospace> and <monospace>dsGPC</monospace> perform similarly, with the former slightly better at classifying remission of hyperactivity, whereas the latter is slightly better for inattentiveness.</p>
<p>Overall for the conventional methods, the <monospace>rwSVC</monospace> achieves the best compromise between SEN and PPV, meaning that it can identify remission cases more readily and at the same time the remission predictions are more reliable. Comparing Tables <xref ref-type="table" rid="T2">2</xref>, <xref ref-type="table" rid="T3">3</xref> it can be seen that the learning in the model space approach is superior overall. With respect to the BAC and AUC measures, the best performing <monospace>BUC</monospace> approach has an advantage of about 6&#x02013;7%. This is interesting given the similar performance in the continuous symptom score prediction task amongst all approaches. The rms error measure in the symptom score prediction task was based on point estimate calculations, and thus used no information on the shape of the posterior distribution. The posterior predictive distribution (Section 3.4) for the <monospace>BUC</monospace> has a Student&#x00027;s <italic>t</italic>-distribution specific to each patient. The distributions were used to construct a probabilistic threshold in trading off specificity and specificity. This subject-specific nonlinear thresholding procedure may have contributed to its performance advantage over other approaches.</p>
</sec>
<sec>
<title>6.2.3. Comparison with literature</title>
<p>As far as the authors are aware, Kim et al. (<xref ref-type="bibr" rid="B36">2015</xref>) is the only published literature on treatment response prediction of ADHD patients using machine learning techniques. Their best attempt achieved an AUC value of 0.84 and 86.4% classification accuracy (that is, the percentage of correct predictions, different from the BAC measure used in this paper) using a wide range of information types including demographical, clinical, genetic, environmental, neuropsychological and neuroimaging measures. In comparison, this paper includes only the more readily obtainable demographical and clinical information and is able to achieve best-case AUCs of 0.82&#x02013;0.84. Restricting to demographical and clinical information, the highest performing method using SVMs in Kim et al. (<xref ref-type="bibr" rid="B36">2015</xref>) had an AUC of 0.69. Granted, the comparison is imprecise because the quality, quantity and sources of demographical and clinical information are different between this paper and Kim et al. (<xref ref-type="bibr" rid="B36">2015</xref>). Judging from the AUC values achieved by SVMs in this paper of about 0.71 (see Table <xref ref-type="table" rid="T3">3</xref>), the results appears to be very close to those in Kim et al. (<xref ref-type="bibr" rid="B36">2015</xref>). Due to this similarity, the previous comparisons should be valid.</p>
</sec>
</sec>
</sec>
<sec id="s7">
<title>7. Clinical utility and further work</title>
<p>The proposed learning in the model space approach is capable of predicting, for an individual, the minimum dosage of a particular medication required to have a user-defined chance of achieving symptomatic remission. It is highly flexible and potentially can be extended to any disease or disorder where medication is used in the course of treatment, speeding up and reducing the cost of the dose optimization/forced titration process, and potentially improving the quality of life for patients by ending the treatment sooner.</p>
<p>The current model, however, does not take into account adverse drug reactions (ADRs), minimization of which is another goal of a dose optimization titration process. To improve clinical utility, it is essential that ADRs are modeled. While data on this are available from the clinical notes accompaning the ADDUCE trial, a different modeling approach is required to incorporate the many different types of ADRs, with prevalence ranging from infrequent to very rare.</p>
<p>While the proposed approach achieves excellent performance in terms of treatment response classification, there is room for improvement. One obvious way to achieve this is to incorporate more data, especially covariates that are functions of time. In this exercise for example, the body mass index and age variables measured at baseline (first appointment) of the patients contribute to the latent factors, which in turn form the baseline variables. As such, they do not vary over time. It may be worth investigating whether the addition of temporal covariates, such as blood pressure, would improve the model.</p>
<p>Another venue for potential improvement is to extend the linear model to a nonlinear model&#x02014;there is no guarantee that all the covariates have a linear relationship with treatment response. Identifying the level and nature of nonlinear relationships is the first challenge. In the current Bayesian framework, the introduction of nonlinearities increases computational complexity for Bayesian inference, requiring the use of techniques such as Gibbs sampling.</p>
<p>There are other areas of interest. For example, what is the optimum strategy, in terms of timing and requirements, for incorporating semi-new patient data to the model space to improve the generalizability of the model for other new patients? How can medical adherence/concordance be modeled? Does gender of the patient play a role in their treatment response?</p>
</sec>
<sec sec-type="conclusions" id="s8">
<title>8. Conclusion</title>
<p>A learning in the model space framework has been utilized to develop a personalized medicine approach to treatment response prediction. First of all, factor analysis was performed to extract latent factors from a large clinical dataset, collected from a UK sample of 157 patients suffering from attention-deficit hyperactivity disorder. The resulting reduced-order patient information was then encoded in a model parameter space resulting in a cloud of personalized models. Then, the patient-specific model space parameters were used to train a Bayesian linear regression model. New patients are then matched to existing patients most similar to themselves to obtain a virtual patient profile, which in turn forms a prior parameter set for the Bayesian linear regression model. Through a Bayesian update algorithm, new data are continuously integrated to improve the prediction performance for a given patient. In addition, the parameters of the &#x0201C;new&#x0201D; patients can be added to the model parameter space (once sufficient data are available) to improve the generalizability of the model for future patients.</p>
<p>Comparisons were made between the learning in the model space approach with conventional data-driven machine learning and regression approaches. In terms of the prediction of the continuous symptom factor scores, the performance of the learning in model space framework was on a par with conventional approaches. However, the new approach is shown to outperform support vector machines, Gaussian processes and linear mixed effects classifiers in the prediction of symptomatic remission. The effective gain in classification performance of the new model can potentially speed up and reduce the cost of a forced titration or dose optimization titration process, which is normally manually performed by the clinician to assess the effective dosage of medication. Further work includes incorporating the prediction of adverse drug reactions, which is also an important element in the dose optimization titration process.</p>
</sec>
<sec id="s9">
<title>Ethics statement</title>
<p>This is secondary data analysis of the ADDUCE ADHD study. ADDUCE has received a favorable ethical opinion from the relevant UK National Health Service (NHS) Research Ethics Committee (REC reference 11/ES/0016). All subjects gave written informed consent in accordance with the Declaration of Helsinki. The protocol was approved by the REC. Consent to information access and secondary data analysis has been granted. All data were handled as per relevant guidelines and rules for such information, including the Data Protection Act 1998 and as per the initial ethical approval.</p>
</sec>
<sec id="s10">
<title>Author contributions</title>
<p>HW was responsible for performing feature extraction of the clinical data, literature search, causal factor model simplification, the software implementation of the learning in model space framework, and determined benchmarking protocol. PAT provided the clinical context of ADHD, constructed the prior knowledge used by the Bayesian framework, aided development on the causal factor model, and designed the literature search protocol. PW aided HW on the literature search and written the background on ADHD. OD implemented the SVM and GP machine learning algorithms. BL constructed the linear mixed effects models. YS helped with the technical implementation of the Bayesian linear regression algorithm. PT provided the mathematical formulation of the learning in model space approach and guidance on software implementation. MC and TN suggested candidate modeling approaches compatible with the framework, suggested a number of conventional approaches to be evaluated, appraised the output of the Bayesian approach, and provided recommendations for the improvement of the implementation and the methodology to ensure fair comparisons. SI coordinated access to the clinical data, facilitated digitization of paper-based clinical records, and maintained the database. DC provided the clinical context of ADHD, suggested causal mechanisms for treatment response, and defined the scope of clinical data collection.</p>
</sec>
<sec id="s11">
<title>Funding</title>
<p>We gratefully acknowledge the support from the UK Engineering and Physical Sciences Research Council (EPSRC), grant number EP/L000296/1. The ADDUCE project from which this piece of research borrowed clinical data is funded by the European Union&#x00027;s Seventh Framework Programme for research, technological development and demonstration under grant agreement number 260576.</p>
<sec>
<title>Conflict of interest statement</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
</sec>
</body>
<back>
<ack>
<p>The authors would like to thank the staff of Ninewells Hospital and Medical School, the University of Dundee, and NHS Tayside; in particular trial database manager Emma McKenzie for coordinating the data exchange and clinical research nurse Jacqueline Paton for her meticulous data digitization efforts. The authors also would like to acknowledge the invaluable feedback by fellow POEMS (Predictive modeling for hEalthcare through MathS) network members and members in the ADDUCE consortium.</p>
</ack>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="http://journal.frontiersin.org/article/10.3389/fphys.2017.00199/full#supplementary-material">http://journal.frontiersin.org/article/10.3389/fphys.2017.00199/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Appendix.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Abdi</surname> <given-names>H.</given-names></name></person-group> (<year>2003</year>). <article-title>Factor rotations</article-title>, in <source>Encyclopedia for Research Methods for the Social Sciences</source>, eds <person-group person-group-type="editor"><name><surname>Lewis-Beck</surname> <given-names>M.</given-names></name> <name><surname>Bryman</surname> <given-names>A.</given-names></name> <name><surname>Futimg</surname> <given-names>T.</given-names></name></person-group> (<publisher-loc>Thousand Oaks, CA</publisher-loc>: <publisher-name>Sage Publications</publisher-name>), <fpage>978</fpage>&#x02013;<lpage>982</lpage>.</citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alberg</surname> <given-names>A. J.</given-names></name> <name><surname>Park</surname> <given-names>J. W.</given-names></name> <name><surname>Hager</surname> <given-names>B. W.</given-names></name> <name><surname>Brock</surname> <given-names>M. V.</given-names></name> <name><surname>Diener-West</surname> <given-names>M.</given-names></name></person-group> (<year>2004</year>). <article-title>The use of &#x0201C;overall accuracy&#x0201D; to evaluate the validity of screening or diagnostic tests</article-title>. <source>J. Gen. Intern. Med.</source> <volume>19</volume>, <fpage>460</fpage>&#x02013;<lpage>465</lpage>. <pub-id pub-id-type="doi">10.1111/j.1525-1497.2004.30091.x</pub-id><pub-id pub-id-type="pmid">15109345</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><collab>American Psychiatric Association</collab></person-group> (<year>2000</year>). <source>Diagnostic and Statistical Manual of Mental Disorders, 4th Edn.</source> <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>APA</publisher-name>.</citation>
</ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><collab>American Psychiatric Association</collab></person-group> (<year>2013</year>). <source>Diagnostic and Statistical Manual of Mental Disorders, 5th Edn</source>. <publisher-loc>Washington, DC</publisher-loc>: <publisher-name>APA</publisher-name>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Angold</surname> <given-names>A.</given-names></name> <name><surname>Costello</surname> <given-names>E. J.</given-names></name> <name><surname>van K&#x000E4;mmen</surname> <given-names>W.</given-names></name> <name><surname>Stouthamer-Loeber</surname> <given-names>M.</given-names></name></person-group> (<year>1996</year>). <article-title>Development of a short questionnaire for use in epidemiological studies of depression in children and adolescents: factor composition and structure across development</article-title>. <source>Int. J. Methods Psychiatr. Res.</source> <volume>5</volume>, <fpage>251</fpage>&#x02013;<lpage>262</lpage>.</citation>
</ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><collab>APS Group Scotland</collab></person-group> (<year>2012</year>). <source>Scottish Index of Multiple Deprivation.</source> Technical Report, The Scottish Government, <publisher-loc>Edinburgh</publisher-loc>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Askland</surname> <given-names>K. D.</given-names></name> <name><surname>Garnaat</surname> <given-names>S.</given-names></name> <name><surname>Sibrava</surname> <given-names>N. J.</given-names></name> <name><surname>Boisseau</surname> <given-names>C. L.</given-names></name> <name><surname>Strong</surname> <given-names>D.</given-names></name> <name><surname>Mancebo</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Prediction of remission in obsessive compulsive disorder using a novel machine learning strategy</article-title>. <source>Int. J. Methods Psychiatr. Res.</source> <volume>24</volume>, <fpage>156</fpage>&#x02013;<lpage>169</lpage>. <pub-id pub-id-type="doi">10.1002/mpr.1463</pub-id><pub-id pub-id-type="pmid">25994109</pub-id></citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Asparouhov</surname> <given-names>T.</given-names></name> <name><surname>Muth&#x000E9;n</surname> <given-names>B.</given-names></name></person-group> (<year>2009</year>). <article-title>Exploratory structural equation modeling</article-title>. <source>Struct. Equat. Model.</source> <volume>16</volume>, <fpage>397</fpage>&#x02013;<lpage>438</lpage>. <pub-id pub-id-type="doi">10.1080/10705510903008204</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Atkins</surname> <given-names>M. S.</given-names></name> <name><surname>Pelham</surname> <given-names>W. E.</given-names></name> <name><surname>Licht</surname> <given-names>M. H.</given-names></name></person-group> (<year>1985</year>). <article-title>A comparison of objective classroom measures and teacher ratings of attention deficit disorder</article-title>. <source>J. Abnorm. Child Psychol.</source> <volume>13</volume>, <fpage>155</fpage>&#x02013;<lpage>167</lpage>. <pub-id pub-id-type="pmid">3973249</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Banaschewski</surname> <given-names>T.</given-names></name> <name><surname>Coghill</surname> <given-names>D.</given-names></name> <name><surname>Santosh</surname> <given-names>P.</given-names></name> <name><surname>Zuddas</surname> <given-names>A.</given-names></name> <name><surname>Asherson</surname> <given-names>P.</given-names></name> <name><surname>Buitelaar</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2006</year>). <article-title>Long-acting medications for the hyperkinetic disorders. A systematic review and European treatment guideline</article-title>. <source>Eur. Child Adolesc. Psychiatry</source> <volume>15</volume>, <fpage>476</fpage>&#x02013;<lpage>495</lpage>. <pub-id pub-id-type="doi">10.1007/s00787-006-0549-0</pub-id><pub-id pub-id-type="pmid">16680409</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barbaresi</surname> <given-names>W. J.</given-names></name> <name><surname>Katusic</surname> <given-names>S. K.</given-names></name> <name><surname>Colligan</surname> <given-names>R. C.</given-names></name> <name><surname>Weaver</surname> <given-names>A. L.</given-names></name> <name><surname>Leibson</surname> <given-names>C. L.</given-names></name> <name><surname>Jacobsen</surname> <given-names>S. J.</given-names></name></person-group> (<year>2006</year>). <article-title>Long-term stimulant medication treatment of attention-deficit/hyperactivity disorder: results from a population-based study</article-title>. <source>J. Dev. Behav. Pediatr.</source> <volume>27</volume>, <fpage>1</fpage>&#x02013;<lpage>10</lpage>. <pub-id pub-id-type="doi">10.1097/00004703-200602000-00001</pub-id><pub-id pub-id-type="pmid">16511362</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bates</surname> <given-names>D.</given-names></name> <name><surname>M&#x000E4;chler</surname> <given-names>M.</given-names></name> <name><surname>Bolker</surname> <given-names>B.</given-names></name> <name><surname>Walker</surname> <given-names>S.</given-names></name></person-group> (<year>2015</year>). <article-title>Fitting linear mixed-effects models using lme4</article-title>. <source>J. Stat. Softw.</source> <volume>67</volume>, <fpage>1</fpage>&#x02013;<lpage>48</lpage>. <pub-id pub-id-type="doi">10.18637/jss.v067.i01</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Bishop</surname> <given-names>C. M.</given-names></name></person-group> (<year>2006</year>). <source>Pattern Recognition and Machine Learning</source>. <publisher-loc>Berlin</publisher-loc>: <publisher-name>Springer-Verlag</publisher-name>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bradley</surname> <given-names>A. P.</given-names></name></person-group> (<year>1997</year>). <article-title>The use of the area under the roc curve in the evaluation of machine learning algorithms</article-title>. <source>Pattern Recogn.</source> <volume>30</volume>, <fpage>1145</fpage>&#x02013;<lpage>1159</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Brodersen</surname> <given-names>K. H.</given-names></name> <name><surname>Ong</surname> <given-names>C. S.</given-names></name> <name><surname>Stephan</surname> <given-names>K. E.</given-names></name> <name><surname>Buhmann</surname> <given-names>J. M.</given-names></name></person-group> (<year>2010</year>). <article-title>The balanced accuracy and its posterior distribution</article-title>, in <source>Pattern Recognition (ICPR), 20th International Conference on</source> (<publisher-loc>Istanbul</publisher-loc>), <fpage>3121</fpage>&#x02013;<lpage>3124</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brodersen</surname> <given-names>K. H.</given-names></name> <name><surname>Schofield</surname> <given-names>T. M.</given-names></name> <name><surname>Leff</surname> <given-names>A. P.</given-names></name> <name><surname>Ong</surname> <given-names>C. S.</given-names></name> <name><surname>Lomakina</surname> <given-names>E. I.</given-names></name> <name><surname>Buhmann</surname> <given-names>J. M.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Generative embedding for model-based classification of fMRI data</article-title>. <source>PLoS Comput. Biol.</source> <volume>7</volume>:<fpage>e1002079</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1002079</pub-id><pub-id pub-id-type="pmid">21731479</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Burges</surname> <given-names>C. J.</given-names></name></person-group> (<year>1998</year>). <article-title>A tutorial on support vector machines for pattern recognition</article-title>. <source>Data Min. Knowl. Discov.</source> <volume>2</volume>, <fpage>121</fpage>&#x02013;<lpage>167</lpage>. <pub-id pub-id-type="doi">10.1023/A:1009715923555</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bussing</surname> <given-names>R.</given-names></name> <name><surname>Fernandez</surname> <given-names>M.</given-names></name> <name><surname>Harwood</surname> <given-names>M.</given-names></name> <name><surname>Hou</surname> <given-names>W.</given-names></name> <name><surname>Garvan</surname> <given-names>C. W.</given-names></name> <name><surname>Swanson</surname> <given-names>J. M.</given-names></name> <etal/></person-group>. (<year>2008</year>). <article-title>Parent and teacher SNAP-IV ratings of attention deficit/hyperactivity disorder symptoms: Psychometric properties and normative ratings from a school district sample</article-title>. <source>Assessment</source> <volume>15</volume>, <fpage>317</fpage>&#x02013;<lpage>328</lpage>. <pub-id pub-id-type="doi">10.1177/1073191107313888</pub-id><pub-id pub-id-type="pmid">18310593</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>C.-C.</given-names></name> <name><surname>Lin</surname> <given-names>C.-J.</given-names></name></person-group> (<year>2011</year>). <article-title>LIBSVM: a library for support vector machines</article-title>. <source>ACM Trans. Intell. Syst. Technol.</source> <volume>2</volume>, <fpage>27</fpage>. <pub-id pub-id-type="doi">10.1145/1961189.1961199</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Ti&#x000F1;o</surname> <given-names>P.</given-names></name> <name><surname>Rodan</surname> <given-names>A.</given-names></name> <name><surname>Yao</surname> <given-names>X.</given-names></name></person-group> (<year>2014</year>). <article-title>Learning in the model space for cognitive fault diagnosis</article-title>. <source>IEEE Trans. Neural Netw. Learn. Syst.</source> <volume>25</volume>, <fpage>124</fpage>&#x02013;<lpage>136</lpage>. <pub-id pub-id-type="doi">10.1109/TNNLS.2013.2256797</pub-id><pub-id pub-id-type="pmid">24806649</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chou</surname> <given-names>W.-J.</given-names></name> <name><surname>Chen</surname> <given-names>S.-J.</given-names></name> <name><surname>Chen</surname> <given-names>Y.-S.</given-names></name> <name><surname>Liang</surname> <given-names>H.-Y.</given-names></name> <name><surname>Lin</surname> <given-names>C.-C.</given-names></name> <name><surname>Tang</surname> <given-names>C.-S.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Remission in children and adolescents diagnosed with attention-deficit/hyperactivity disorder via an effective and tolerable titration scheme for osmotic release oral system methylphenidate</article-title>. <source>J. Child Adolesc. Psychopharmacol.</source> <volume>22</volume>, <fpage>215</fpage>&#x02013;<lpage>225</lpage>. <pub-id pub-id-type="doi">10.1089/cap.2011.0006</pub-id><pub-id pub-id-type="pmid">22537358</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dopheide</surname> <given-names>J. A.</given-names></name> <name><surname>Pliszka</surname> <given-names>S. R.</given-names></name></person-group> (<year>2009</year>). <article-title>Attention-deficit-hyperactivity disorder: an update</article-title>. <source>Pharmacotherapy</source> <volume>29</volume>, <fpage>656</fpage>&#x02013;<lpage>679</lpage>. <pub-id pub-id-type="doi">10.1592/phco.29.6.656</pub-id><pub-id pub-id-type="pmid">19476419</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Doyle</surname> <given-names>O. M.</given-names></name> <name><surname>Tsaneva-Atansaova</surname> <given-names>K.</given-names></name> <name><surname>Harte</surname> <given-names>J.</given-names></name> <name><surname>Tiffin</surname> <given-names>P. A.</given-names></name> <name><surname>Ti&#x000F1;o</surname> <given-names>P.</given-names></name> <name><surname>Diaz-Zuccarini</surname> <given-names>V.</given-names></name></person-group> (<year>2013</year>). <article-title>Bridging paradigms: hybrid mechanistic-discriminative predictive models</article-title>. <source>IEEE Trans. Biomed. Eng.</source> <volume>60</volume>, <fpage>735</fpage>&#x02013;<lpage>742</lpage>. <pub-id pub-id-type="doi">10.1109/TBME.2013.2244598</pub-id><pub-id pub-id-type="pmid">23392334</pub-id></citation>
</ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fawcett</surname> <given-names>T.</given-names></name></person-group> (<year>2006</year>). <article-title>An introduction to roc analysis</article-title>. <source>Pattern Recogn. Lett.</source> <volume>27</volume>, <fpage>861</fpage>&#x02013;<lpage>874</lpage>. <pub-id pub-id-type="doi">10.1016/j.patrec.2005.10.010</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Fletcher</surname> <given-names>R.</given-names></name> <name><surname>Fletcher</surname> <given-names>S.</given-names></name></person-group> (<year>2005</year>). <source>Clinical Epidemiology: The Essentials</source>. Epidemiology/Biostatistics. <publisher-loc>Baltimore, MD</publisher-loc>: <publisher-name>Lippincott Williams &#x00026; Wilkins</publisher-name>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goodman</surname> <given-names>R.</given-names></name></person-group> (<year>1997-07</year>). <article-title>The strengths difficulties questionnaire: a research note</article-title>. <source>J. Child Psychol. Psychiatry</source> <volume>38</volume>, <fpage>581</fpage>&#x02013;<lpage>586</lpage>. <pub-id pub-id-type="doi">10.1111/j.1469-7610.1997.tb01545.x</pub-id><pub-id pub-id-type="pmid">9255702</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goodman</surname> <given-names>R.</given-names></name> <name><surname>Ford</surname> <given-names>T.</given-names></name> <name><surname>Richards</surname> <given-names>H.</given-names></name> <name><surname>Gatward</surname> <given-names>R.</given-names></name> <name><surname>Meltzer</surname> <given-names>H.</given-names></name></person-group> (<year>2000</year>). <article-title>The development and well-being assessment: description and initial validation of an integrated assessment of child and adolescent psychopathology</article-title>. <source>J. Child Psychol. Psychiatry</source> <volume>41</volume>, <fpage>645</fpage>&#x02013;<lpage>655</lpage>. <pub-id pub-id-type="doi">10.1111/j.1469-7610.2000.tb02345.x</pub-id><pub-id pub-id-type="pmid">10946756</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Goodfellow</surname> <given-names>I.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Courville</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <source>Deep Learning</source>. <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>MIT Press</publisher-name>.</citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Greenhill</surname> <given-names>L.</given-names></name> <name><surname>Kollins</surname> <given-names>S.</given-names></name> <name><surname>Abikoff</surname> <given-names>H.</given-names></name> <name><surname>McCracken</surname> <given-names>J.</given-names></name> <name><surname>Riddle</surname> <given-names>M.</given-names></name> <name><surname>Swanson</surname> <given-names>J. M.</given-names></name></person-group> (<year>2006</year>). <article-title>Efficacy and safety of immediate-release methylphenidate treatment for preschoolers with ADHD</article-title>. <source>J. Am. Acad. Child Adolesc. Psychiatry</source> <volume>45</volume>, <fpage>1284</fpage>&#x02013;<lpage>1293</lpage>. <pub-id pub-id-type="doi">10.1097/01.chi.0000235077.32661.61</pub-id><pub-id pub-id-type="pmid">17023867</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Guy</surname> <given-names>W.</given-names></name></person-group> (<year>1974</year>). <source>ECDEU Assessment Manual for Psychopharmacology, Revised Edn.</source> <publisher-loc>Rockville, MD</publisher-loc>: <publisher-name>US Department of Health, Education and Welfare, Public Health Service, Alcohol, Drug Abuse and Mental Health Administration, NIMH Psychopharmacology Research Branch, Division of Extramural Research Programs</publisher-name>. DHEW No. ADM 76-338.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>H.</given-names></name> <name><surname>Garcia</surname> <given-names>E.</given-names></name></person-group> (<year>2009</year>). <article-title>Learning from imbalanced data</article-title>. <source>IEEE Trans. Knowledge Data Eng.</source> <volume>21</volume>, <fpage>1263</fpage>&#x02013;<lpage>1284</lpage>. <pub-id pub-id-type="doi">10.1109/TKDE.2008.239</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hechtman</surname> <given-names>L.</given-names></name></person-group> (<year>2005</year>). <article-title>Effects of treatment on the overall functioning of children with ADHD</article-title>. <source>Can. Child Adolesc. Psychiatr. Rev.</source> <volume>14</volume>, <fpage>10</fpage>&#x02013;<lpage>15</lpage>. <pub-id pub-id-type="pmid">19030519</pub-id></citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Holden</surname> <given-names>S. E.</given-names></name> <name><surname>Jenkins-Jones</surname> <given-names>S.</given-names></name> <name><surname>Poole</surname> <given-names>C. D.</given-names></name> <name><surname>Morgan</surname> <given-names>C. L.</given-names></name> <name><surname>Coghill</surname> <given-names>D.</given-names></name> <name><surname>Currie</surname> <given-names>C. J.</given-names></name></person-group> (<year>2013</year>). <article-title>The prevalence and incidence, resource use and financial costs of treating people with attention deficit/hyperactivity disorder (ADHD) in the United Kingdom (1998 to 2010)</article-title>. <source>Child Adolesc. Psychiatry Ment. Health</source> <volume>7</volume>:<fpage>34</fpage>. <pub-id pub-id-type="doi">10.1186/1753-2000-7-34</pub-id><pub-id pub-id-type="pmid">24119376</pub-id></citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Horn</surname> <given-names>J. L.</given-names></name></person-group> (<year>1965</year>). <article-title>A rationale and test for the number of factors in factor analysis</article-title>. <source>Psychometrika</source> <volume>30</volume>, <fpage>179</fpage>&#x02013;<lpage>185</lpage>. <pub-id pub-id-type="doi">10.1007/BF02289447</pub-id><pub-id pub-id-type="pmid">14306381</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khangura</surname> <given-names>S.</given-names></name> <name><surname>Konnyu</surname> <given-names>K.</given-names></name> <name><surname>Cushman</surname> <given-names>R.</given-names></name> <name><surname>Grimshaw</surname> <given-names>J.</given-names></name> <name><surname>Moher</surname> <given-names>D.</given-names></name></person-group> (<year>2012</year>). <article-title>Evidence summaries: the evolution of a rapid review approach</article-title>. <source>Syst. Rev.</source> <volume>1</volume>:<fpage>10</fpage>. <pub-id pub-id-type="doi">10.1186/2046-4053-1-10</pub-id><pub-id pub-id-type="pmid">22587960</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>J.</given-names></name> <name><surname>Sharma</surname> <given-names>V.</given-names></name> <name><surname>Ryan</surname> <given-names>N. D.</given-names></name></person-group> (<year>2015</year>). <article-title>Predicting methylphenidate response in ADHD using machine learning approaches</article-title>. <source>Int. J. Neuropsychopharmacol.</source> <volume>18</volume>, <fpage>1</fpage>&#x02013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1093/ijnp/pyv052</pub-id><pub-id pub-id-type="pmid">25964505</pub-id></citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kliegl</surname> <given-names>R.</given-names></name> <name><surname>Rolfs</surname> <given-names>M.</given-names></name> <name><surname>Laubrock</surname> <given-names>J.</given-names></name> <name><surname>Engbert</surname> <given-names>R.</given-names></name></person-group> (<year>2009</year>). <article-title>Microsaccadic modulation of response times in spatial attention tasks</article-title>. <source>Psychol. Res.</source> <volume>73</volume>, <fpage>136</fpage>&#x02013;<lpage>146</lpage>. <pub-id pub-id-type="doi">10.1007/s00426-008-0202-2</pub-id><pub-id pub-id-type="pmid">19066951</pub-id></citation>
</ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leckman</surname> <given-names>J. F.</given-names></name> <name><surname>Riddle</surname> <given-names>M. A.</given-names></name> <name><surname>Hardin</surname> <given-names>M. T.</given-names></name> <name><surname>Ort</surname> <given-names>S. I.</given-names></name> <name><surname>Swartz</surname> <given-names>K. L.</given-names></name> <name><surname>Stevenson</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>1989</year>). <article-title>The Yale Global Tic Severity Scale: initial testing of a clinician-rated scale of tic severity</article-title>. <source>J. Am. Acad. Child Adolesc. Psychiatry</source> <volume>28</volume>, <fpage>566</fpage>&#x02013;<lpage>573</lpage>. <pub-id pub-id-type="doi">10.1097/00004583-198907000-00015</pub-id><pub-id pub-id-type="pmid">2768151</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lorenzo-Seva</surname> <given-names>U.</given-names></name> <name><surname>Ferrando</surname> <given-names>P. J.</given-names></name></person-group> (<year>2006</year>). <article-title>FACTOR: a computer program to fit the exploratory factor analysis model</article-title>. <source>Behav. Res. Methods</source> <volume>38</volume>, <fpage>88</fpage>&#x02013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.3758/BF03192753</pub-id><pub-id pub-id-type="pmid">16817517</pub-id></citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McCarthy</surname> <given-names>S.</given-names></name> <name><surname>Asherson</surname> <given-names>P.</given-names></name> <name><surname>Coghill</surname> <given-names>D.</given-names></name> <name><surname>Hollis</surname> <given-names>C.</given-names></name> <name><surname>Murray</surname> <given-names>M.</given-names></name> <name><surname>Potts</surname> <given-names>L.</given-names></name></person-group> (<year>2009</year>). <article-title>Attention-deficit hyperactivity disorder: treatment discontinuation in adolescents and young adults</article-title>. <source>Br. J. Psychiatry</source> <fpage>273</fpage>&#x02013;<lpage>277</lpage>. <pub-id pub-id-type="doi">10.1192/bjp.bp.107.045245</pub-id><pub-id pub-id-type="pmid">19252159</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Muth&#x000E9;n</surname> <given-names>B. O.</given-names></name> <name><surname>du Toit</surname> <given-names>S. H. C.</given-names></name> <name><surname>Spisic</surname> <given-names>D.</given-names></name></person-group> (<year>1997</year>). <source>Robust Inference using Weighted Least Squares and Quadratic Estimating Equations in Latent Variable Modeling with Categorical and Continuous Outcomes</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.statmodel.com/download/Article_075.pdf">https://www.statmodel.com/download/Article_075.pdf</ext-link></citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nooteboom</surname> <given-names>S. G.</given-names></name> <name><surname>Quen&#x000E9;</surname> <given-names>H.</given-names></name></person-group> (<year>2008</year>). <article-title>Self-monitoring and feedback: a new attempt to find the main cause of lexical bias in phonological speech errors</article-title>. <source>J. Mem. Lang.</source> <volume>58</volume>, <fpage>837</fpage>&#x02013;<lpage>861</lpage>. <pub-id pub-id-type="doi">10.1016/j.jml.2007.05.003</pub-id></citation>
</ref>
<ref id="B43">
<citation citation-type="book"><person-group person-group-type="editor"><name><surname>O&#x00027;Hagan</surname> <given-names>A.</given-names></name> <name><surname>Forester</surname> <given-names>J. J.</given-names></name></person-group> (eds.). (<year>2004</year>). <article-title>The linear model</article-title>, in <source>Bayesian Inference, Vol. 2b, Kendall&#x00027;s Advanced Theory of Statistics</source>, <edition>2nd Edn.</edition> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>John Wiley &#x00026; Sons</publisher-name>), <fpage>305</fpage>&#x02013;<lpage>339</lpage>.</citation>
</ref>
<ref id="B44">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Osuna</surname> <given-names>E.</given-names></name> <name><surname>Freund</surname> <given-names>R.</given-names></name> <name><surname>Girosi</surname> <given-names>F.</given-names></name></person-group> (<year>1997</year>). <source>Support Vector Machines: Training and Applications.</source> Technical Report AIM-1602, <publisher-name>Massachusetts Institute of Technology</publisher-name>, <publisher-loc>Cambridge, MA</publisher-loc>.</citation>
</ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Powers</surname> <given-names>D. M. W.</given-names></name></person-group> (<year>2011</year>). <article-title>Evaluation: from precision, recall and <italic>F</italic>-measure to ROC, informedness, markedness &#x00026; correlation</article-title>. <source>J. Mach. Learn. Technol.</source> <volume>2</volume>, <fpage>37</fpage>&#x02013;<lpage>63</lpage>. <pub-id pub-id-type="doi">10.9735/2229-3981</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="book"><person-group person-group-type="author"><collab>R Core Team</collab></person-group> (<year>2013</year>). <source>R: A Language and Environment for Statistical Computing</source>. <publisher-loc>Vienna</publisher-loc>: <publisher-name>R Foundation for Statistical Computing</publisher-name>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rasmussen</surname> <given-names>C. E.</given-names></name> <name><surname>Nickisch</surname> <given-names>H.</given-names></name></person-group> (<year>2010</year>). <article-title>Gaussian processes for machine learning (GPML) toolbox</article-title>. <source>J. Mach. Learn. Res.</source> <volume>11</volume>, <fpage>3011</fpage>&#x02013;<lpage>3015</lpage>.</citation>
</ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Reeves</surname> <given-names>G.</given-names></name> <name><surname>Schweitzer</surname> <given-names>J.</given-names></name></person-group> (<year>2004</year>). <article-title>Pharmacological management of attention-deficit hyperactivity disorder</article-title>. <source>Expert Opin. Pharmacother.</source> <volume>5</volume>, <fpage>1313</fpage>&#x02013;<lpage>1320</lpage>. <pub-id pub-id-type="doi">10.1517/14656566.5.6.1313</pub-id><pub-id pub-id-type="pmid">15163276</pub-id></citation>
</ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rico</surname> <given-names>D.</given-names></name> <name><surname>Martin-Diana</surname> <given-names>A. B.</given-names></name> <name><surname>Bary-Ryan</surname> <given-names>C.</given-names></name> <name><surname>Henehan</surname> <given-names>G. T. M.</given-names></name> <name><surname>Frias</surname> <given-names>J. M.</given-names></name></person-group> (<year>2007</year>). <article-title>Simultaneous modelling of the thermal degradation kinetics of pectin methylesterase in lettuce (<italic>Lactuca sativa</italic> L.) and carrot (<italic>Daucus carota</italic> L.) extracts: analysis of seasonal variation and tissue type</article-title>. <source>Biosci. Biotechnol. Biochem.</source> <volume>71</volume>, <fpage>2383</fpage>&#x02013;<lpage>2392</lpage>. <pub-id pub-id-type="doi">10.1271/bbb.60484</pub-id><pub-id pub-id-type="pmid">17928712</pub-id></citation>
</ref>
<ref id="B50">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rutter</surname> <given-names>M.</given-names></name> <name><surname>Bailey</surname> <given-names>A.</given-names></name> <name><surname>Lord</surname> <given-names>C.</given-names></name></person-group> (<year>2003</year>). <source>SCQ: The Social Communication Questionnaire</source>. <publisher-loc>Torrance, CA</publisher-loc>: <publisher-name>Western Psychological Services</publisher-name>.</citation>
</ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Safer</surname> <given-names>D. J.</given-names></name> <name><surname>Zito</surname> <given-names>J. M.</given-names></name> <name><surname>Fine</surname> <given-names>E. M.</given-names></name></person-group> (<year>1996</year>). <article-title>Increased methylphenidate usage for attention deficit disorder in the 1990s</article-title>. <source>Pediatrics</source> <volume>98</volume>, <fpage>1084</fpage>&#x02013;<lpage>1088</lpage>. <pub-id pub-id-type="pmid">8951257</pub-id></citation>
</ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schachter</surname> <given-names>H. M.</given-names></name> <name><surname>Pham</surname> <given-names>B.</given-names></name> <name><surname>King</surname> <given-names>J.</given-names></name> <name><surname>Langford</surname> <given-names>S.</given-names></name> <name><surname>Moher</surname> <given-names>D.</given-names></name></person-group> (<year>2001</year>). <article-title>How efficacious and safe is short-acting methylphenidate for the treatment of attention-deficit disorder in children and adolescents? A meta-analysis</article-title>. <source>Can. Med. Assoc. J.</source> <volume>165</volume>, <fpage>1475</fpage>&#x02013;<lpage>1488</lpage>. <pub-id pub-id-type="pmid">11762571</pub-id></citation>
</ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schulz</surname> <given-names>K. F.</given-names></name> <name><surname>Altman</surname> <given-names>D. G.</given-names></name> <name><surname>Moher</surname> <given-names>D.</given-names></name></person-group> (<year>2010</year>). <article-title>CONSORT 2010 statement: updated guidelines for reporting parallel group randomised trials</article-title>. <source>BMJ</source> <volume>340</volume>:<fpage>c332</fpage>. <pub-id pub-id-type="doi">10.1136/bmj.c332</pub-id><pub-id pub-id-type="pmid">20332509</pub-id></citation>
</ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shaffer</surname> <given-names>D.</given-names></name> <name><surname>Gould</surname> <given-names>M. S.</given-names></name> <name><surname>Brasic</surname> <given-names>J.</given-names></name> <name><surname>Ambrosini</surname> <given-names>P.</given-names></name> <name><surname>Fisher</surname> <given-names>P.</given-names></name> <name><surname>Bird</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>1983</year>). <article-title>A children&#x00027;s global assessment scale (CGAS)</article-title>. <source>Arch. Gen. Psychiatry</source> <volume>40</volume>, <fpage>1228</fpage>. <pub-id pub-id-type="doi">10.1001/archpsyc.1983.01790100074010</pub-id><pub-id pub-id-type="pmid">6639293</pub-id></citation>
</ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname> <given-names>Y.</given-names></name> <name><surname>Ti&#x000F1;o</surname> <given-names>P.</given-names></name> <name><surname>Tsaneva-Atanasova</surname> <given-names>K.</given-names></name></person-group> (<year>2016</year>). <article-title>A classification framework for partially observed dynamical systems</article-title>. <source>ArXiv e-prints</source>.</citation>
</ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Spencer</surname> <given-names>T.</given-names></name> <name><surname>Biederman</surname> <given-names>J.</given-names></name> <name><surname>Wilens</surname> <given-names>T.</given-names></name> <name><surname>Harding</surname> <given-names>M.</given-names></name> <name><surname>O&#x00027;Donnell</surname> <given-names>D.</given-names></name> <name><surname>Griffin</surname> <given-names>S.</given-names></name></person-group> (<year>1996</year>). <article-title>Pharmacotherapy of attention-deficit hyperactivity disorder across the life cycle</article-title>. <source>J. Am. Acad. Child Adolesc. Psychiatr.</source> <volume>35</volume>, <fpage>409</fpage>&#x02013;<lpage>432</lpage>. <pub-id pub-id-type="doi">10.1097/00004583-199604000-00008</pub-id><pub-id pub-id-type="pmid">8919704</pub-id></citation>
</ref>
<ref id="B57">
<citation citation-type="book"><person-group person-group-type="author"><collab>StataCorp</collab></person-group> (<year>2015</year>). <source>Stata Statistical Software: Release 14</source>. <publisher-name>College Station, TX</publisher-name>:<publisher-name>StataCorp LP</publisher-name>.</citation>
</ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stevens</surname> <given-names>M. H. H.</given-names></name> <name><surname>Sanchez</surname> <given-names>M.</given-names></name> <name><surname>Lee</surname> <given-names>J.</given-names></name> <name><surname>Finkel</surname> <given-names>S. E.</given-names></name></person-group> (<year>2007</year>). <article-title>Diversification rates increase with population size and resource concentration in an unstructured habitat</article-title>. <source>Genetics</source> <volume>177</volume>, <fpage>2243</fpage>&#x02013;<lpage>2250</lpage>. <pub-id pub-id-type="doi">10.1534/genetics.107.076869</pub-id><pub-id pub-id-type="pmid">18073429</pub-id></citation>
</ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Storeb&#x000F8;</surname> <given-names>O. J.</given-names></name> <name><surname>Ramstad</surname> <given-names>E.</given-names></name> <name><surname>Krogh</surname> <given-names>H. B.</given-names></name> <name><surname>Nilausen</surname> <given-names>T. D.</given-names></name> <name><surname>Skoog</surname> <given-names>M.</given-names></name> <name><surname>Holmskov</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Methylphenidate for children and adolescents with attention deficit hyperactivity disorder (ADHD)</article-title>. <source>Cochrane Database Syst. Rev.</source> <volume>11</volume>:<fpage>CD009885</fpage>. <pub-id pub-id-type="doi">10.1002/14651858.CD009885.pub2</pub-id></citation>
</ref>
<ref id="B60">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Swanson</surname> <given-names>J. M.</given-names></name></person-group> (<year>1992</year>). <source>School-Based Assessments and Interventions for ADD Students</source>. <publisher-loc>Irvine, CA</publisher-loc>: <publisher-name>KC Publishing</publisher-name>.</citation>
</ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Swanson</surname> <given-names>J. M.</given-names></name> <name><surname>Kraemer</surname> <given-names>H. C.</given-names></name> <name><surname>Hinshaw</surname> <given-names>S. P.</given-names></name> <name><surname>Arnold</surname> <given-names>L. E.</given-names></name> <name><surname>Conners</surname> <given-names>C. K.</given-names></name> <name><surname>Abikoff</surname> <given-names>H. B.</given-names></name></person-group> (<year>2001</year>). <article-title>Clinical relevance of the primary findings of the MTA: success rates based on severity of ADHD and ODD symptoms at the end of treatment</article-title>. <source>J. Am. Acad. Child Adolesc.</source> <volume>40</volume>, <fpage>168</fpage>&#x02013;<lpage>179</lpage>. <pub-id pub-id-type="doi">10.1097/00004583-200102000-00011</pub-id><pub-id pub-id-type="pmid">11211365</pub-id></citation>
</ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Swanson</surname> <given-names>J. M.</given-names></name> <name><surname>Sandman</surname> <given-names>C. A.</given-names></name> <name><surname>Deutsch</surname> <given-names>C.</given-names></name> <name><surname>Baren</surname> <given-names>M.</given-names></name></person-group> (<year>1983</year>). <article-title>Methylphenidate hydrochloride given with or before breakfast: I. Behavioral, cognitive, and electrophysiologic effects</article-title>. <source>Pediatrics</source> <volume>72</volume>, <fpage>49</fpage>&#x02013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1097/00004583-198311000-00019</pub-id><pub-id pub-id-type="pmid">6866591</pub-id></citation>
</ref>
<ref id="B63">
<citation citation-type="web"><person-group person-group-type="author"><collab>The ADDUCE Consortium</collab></person-group> (<year>2016</year>). <source>The ADDUCE (Attention Deficit/Hyperactivity Disorder Drugs Use Chronic Effects) Project</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.adhd-adduce.org/">http://www.adhd-adduce.org/</ext-link></citation>
</ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thurstone</surname> <given-names>L. L.</given-names></name></person-group> (<year>1931</year>). <article-title>Multiple factor analysis</article-title>. <source>Psychol. Rev.</source> <volume>38</volume>, <fpage>406</fpage>&#x02013;<lpage>427</lpage>. <pub-id pub-id-type="doi">10.1037/h0069792</pub-id></citation>
</ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>van der Oord</surname> <given-names>S.</given-names></name> <name><surname>Prins</surname> <given-names>P. J. M.</given-names></name> <name><surname>Oosterlaan</surname> <given-names>J.</given-names></name> <name><surname>Emmelkamp</surname> <given-names>P. M. G.</given-names></name></person-group> (<year>2008</year>). <article-title>Efficacy of methylphenidate, psychosocial treatments and their combination in school-aged children with ADHD: a meta-analysis</article-title>. <source>Clin. Psychol. Rev.</source> <volume>28</volume>, <fpage>783</fpage>&#x02013;<lpage>800</lpage>. <pub-id pub-id-type="doi">10.1016/j.cpr.2007.10.007</pub-id><pub-id pub-id-type="pmid">18068284</pub-id></citation>
</ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>von Elm</surname> <given-names>E.</given-names></name> <name><surname>Altman</surname> <given-names>D. G.</given-names></name> <name><surname>Egger</surname> <given-names>M.</given-names></name> <name><surname>Pocock</surname> <given-names>S. J.</given-names></name> <name><surname>G&#x000F8;tzsche</surname> <given-names>P. C.</given-names></name> <name><surname>Vandenbroucke</surname> <given-names>J. P.</given-names></name></person-group> (<year>2007</year>). <article-title>Strengthening the reporting of observational studies in epidemiology (STROBE) statement: guidelines for reporting observational studies</article-title>. <source>BMJ</source> <volume>335</volume>, <fpage>806</fpage>&#x02013;<lpage>808</lpage>. <pub-id pub-id-type="doi">10.1136/bmj.39335.541782.AD</pub-id><pub-id pub-id-type="pmid">17947786</pub-id></citation>
</ref>
<ref id="B67">
<citation citation-type="book"><person-group person-group-type="author"><collab>World Health Organization</collab></person-group> (<year>2010</year>). <source>ICD-10, 10 Edn</source>. <publisher-loc>Geneva</publisher-loc>: <publisher-name>World Health Organization</publisher-name>.</citation>
</ref>
<ref id="B68">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Youden</surname> <given-names>W. J.</given-names></name></person-group> (<year>1950</year>). <article-title>Index for rating diagnostic tests</article-title>. <source>Cancer</source> <volume>3</volume>, <fpage>32</fpage>&#x02013;<lpage>35</lpage>. <pub-id pub-id-type="doi">10.1002/1097-0142(1950)3:1&#x0003C;32::AID-CNCR2820030106&#x0003E;3.0.CO;2-3</pub-id><pub-id pub-id-type="pmid">15405679</pub-id></citation>
</ref>
</ref-list>
</back>
</article>