<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="review-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2023.1129380</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Oncology</subject>
<subj-group>
<subject>Review</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>On the importance of interpretable machine learning predictions to inform clinical decision making in oncology</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Lu</surname>
<given-names>Sheng-Chieh</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2146931"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Swisher</surname>
<given-names>Christine L.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2195159"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chung</surname>
<given-names>Caroline</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/623077"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jaffray</surname>
<given-names>David</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Sidey-Gibbons</surname>
<given-names>Chris</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Section of Patient-Centered Analytics, Division of Internal Medicine, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>The Ronin Project</institution>, <addr-line>San Mateo, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>The Lawrence J. Ellison Institute for Transformative Medicine</institution>, <addr-line>Los Angeles, CA</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Department of Radiation Oncology, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Institute for Data Science in Oncology, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Imaging Physics, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Radiation Physics, The University of Texas MD Anderson Cancer Center</institution>, <addr-line>Houston, TX</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Kevin Camphausen, National Cancer Institute, National Institutes of Health (NIH), United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Han Yu, Roswell Park Comprehensive Cancer Center, United States; Andra Valentina Krauze, Center for Cancer Research, National Institutes of Health (NIH), United States</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Chris Sidey-Gibbons, <email xlink:href="mailto:cgibbons@mdanderson.org">cgibbons@mdanderson.org</email>
</p>
</fn>
<fn fn-type="other" id="fn002">
<p>This article was submitted to Radiation Oncology, a section of the journal Frontiers in Oncology</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>02</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>13</volume>
<elocation-id>1129380</elocation-id>
<history>
<date date-type="received">
<day>21</day>
<month>12</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>14</day>
<month>02</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2023 Lu, Swisher, Chung, Jaffray and Sidey-Gibbons</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Lu, Swisher, Chung, Jaffray and Sidey-Gibbons</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Machine learning-based tools are capable of guiding individualized clinical management and decision-making by providing predictions of a patient&#x2019;s future health state. Through their ability to model complex nonlinear relationships, ML algorithms can often outperform traditional statistical prediction approaches, but the use of nonlinear functions can mean that ML techniques may also be less interpretable than traditional statistical methodologies. While there are benefits of intrinsic interpretability, many model-agnostic approaches now exist and can provide insight into the way in which ML systems make decisions. In this paper, we describe how different algorithms can be interpreted and introduce some techniques for interpreting complex nonlinear algorithms.</p>
</abstract>
<kwd-group>
<kwd>opaque machine learning models</kwd>
<kwd>interpretability and explainability</kwd>
<kwd>decision-making support</kwd>
<kwd>high-stakes prediction</kwd>
<kwd>precision medicine</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="4"/>
<equation-count count="0"/>
<ref-count count="72"/>
<page-count count="13"/>
<word-count count="7015"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Machine learning (ML) techniques have demonstrated exceptional promise in producing reliable predictions to inspire action across diverse industries. They have been fundamental in the automation of complex tasks such as language translation, self-driving vehicles, as well as internet search and recommendation engines. In oncology, there are many applications across the care continuum from informing healthcare policy, managing clinical operations, to providing individualized insights into direct patient care (<xref ref-type="bibr" rid="B1">1</xref>&#x2013;<xref ref-type="bibr" rid="B4">4</xref>).</p>
<p>The principle of using data-driven prediction models to inform clinical oncology care is not new, though the increased availability and maturity of the capabilities of ML techniques has led to renewed interest in the topic. Traditional prediction tools tend to be developed using statistical methodologies (<xref ref-type="bibr" rid="B5">5</xref>). For example, oncology nomograms often utilized linear algorithms to create tools with which future outcomes could be predicted. These models were often based on ordinal least squares regression techniques which offered straightforward interpretability using coefficients. However, machine learning&#x2019;s ability to characterize nonlinear interactions between features has led to potential issues with understanding the relationship between the input features and the output prediction. These nonlinear algorithms are often referred to as &#x2018;black boxes&#x2019; which may produce accurate predictions but at the expense of clear and concise interpretability. Although many ML models can be adequately thought of as &#x2018;black boxes&#x2019;, it is not true that all ML algorithms are uninterpretable. In previous work (<xref ref-type="bibr" rid="B6">6</xref>, <xref ref-type="bibr" rid="B7">7</xref>), we have previously described a continuum of algorithms ranging from &#x2018;Auditable Algorithms&#x2019; to &#x2018;Black Boxes&#x2019; and argued that interpretability necessarily became more difficult and the ability to estimate highly complex nonlinear models increased.</p>
<p>In recent years, the machine learning community has produced several significant advancements into providing some level of interpretability for complex nonlinear algorithms (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>). Explainability and interpretability are two tied concepts (<xref ref-type="bibr" rid="B10">10</xref>). There are no clear and widely-accepted definitions of these terms, so we will use a working definition inspired by other sources (<xref ref-type="bibr" rid="B11">11</xref>). Explainability refers to the ability to describe the elements of an ML model, which might include the provenance and nature of the training data, weights, and coefficients of the models, or the importance of different features in deriving the prediction (<xref ref-type="bibr" rid="B10">10</xref>). Explainability asks the question &#x201c;can we <italic>describe</italic> the different elements of the model?&#x201d;. The concept of interpretability goes beyond that of description of explainability and asks &#x201c;can we <italic>understand</italic> the reasoning behind the model&#x2019;s prediction?&#x201d; (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B12">12</xref>). In this paper, we will focus on the description of features and explanation approaches that make an ML model interpretable by allowing humans to gain insight into model reasoning and consistently predict model outputs.</p>
<p>Interpretability is an important concept within clinical ML as model performance is unlikely to be perfect, and the provision of an interpretable explanation can aid in decision-making using ML models. The importance of interpretability for all ML-based decision-making algorithms is demonstrated in the United States Government&#x2019;s Blueprint for an AI Bill of Rights which introduces &#x201c;Notice and Explanation&#x201d; as a key principle for ML-based prediction models (<xref ref-type="bibr" rid="B13">13</xref>). Additionally, the U.S. Food and Drug Administration (FDA) guidelines for clinical decision support systems (CDSS) highlight the importance of providing the basis of predictions (<xref ref-type="bibr" rid="B14">14</xref>), and other regulatory and standards in healthcare and other industries (<xref ref-type="bibr" rid="B15">15</xref>).</p>
<p>For oncology practice, ML-based tools are often developed and used to support high-stakes decisions, such as diagnosis (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B17">17</xref>), advance care planning communication (<xref ref-type="bibr" rid="B18">18</xref>, <xref ref-type="bibr" rid="B19">19</xref>), and treatment selection (<xref ref-type="bibr" rid="B20">20</xref>). Providing only predictions is not enough to solve all problems for these tasks, and a model should provide explanations concerning its decision-making to allow human reasoning and preventative actions (<xref ref-type="bibr" rid="B11">11</xref>). Furthermore, interpretability is essential to ensure safety, ethics, and accountability of the models for ML models supporting oncology decisions (<xref ref-type="bibr" rid="B11">11</xref>). Inaccurate or biased predictions generated by a ML model can result in unintentional harms on both patients and institutions. In such cases, an explanation in model decision-making process pertaining to erroneous or discriminative predictions enables model auditing, debugging, and refinement to ensure model performance and fairness (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B21">21</xref>).</p>
<p>Models making predictions using different types of data should be interpreted by different approaches. For instance, a common approach to interpret ML models leveraging image data is the salience map highlighting a portion of an image that is most relevant to model decisions (<xref ref-type="bibr" rid="B22">22</xref>). Many explanation approaches, such as attention, are also available for the provision of insights into the decision-making processes of models leveraging unstructured text data using natural language processing (<xref ref-type="bibr" rid="B23">23</xref>). As there is increasing enthusiasm for leveraging electronic health record data to construct predictive decision-making tools (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B24">24</xref>, <xref ref-type="bibr" rid="B25">25</xref>), we focus this paper on approaches most useful in deconstructing decision-making processes of opaque ML models using tabular data. Nevertheless, many of the interpretation approaches we covered are not data type constrained (<xref ref-type="bibr" rid="B23">23</xref>, <xref ref-type="bibr" rid="B26">26</xref>).</p>
<p>In this manuscript, we demonstrate how interpretability and explainability of machine learning models can be informed both by algorithm selection and the applications of so-called &#x201c;model agnostic&#x201d; methods at the population and individual levels (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B9">9</xref>). We also describe several of the benefits and limitations of both intrinsic interpretability such as that provided with logistic regression versus model agnostic methods. Additionally, we argue that interpretability can go beyond the drivers of an individual prediction and may also encompass methods to understand the quality, relevance, and distributions of training, testing, and inference data features which we used to inform the model. The intention of this manuscript was not to provide an exhaustive summary of state-of-the-art ML interpretation approaches but to introduce the concept of ML interpretability and explainability with practical examples to raise awareness of the topic among the oncology research community. For enthusiastic readers, there are systematic reviews that provide more compressive summaries of the existing model interpretation techniques (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B27">27</xref>, <xref ref-type="bibr" rid="B28">28</xref>).</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Example models used to illustrate explanation methods for interpretability</title>
<p>In this paper, we demonstrate all model interpretation approaches with example models we created to identify cancerous breast masses using regularized linear regression (GLM), multivariate adaptive regression splines (MARS), k-Nearest Neighbors, Decision trees, extreme gradient boosting (XGB), and neural networks (NNET). We used the Breast Cancer Wisconsin Diagnostic Data Set which is publicly available from the University of California Irvine (UCI) ML Repository to train the models (<xref ref-type="bibr" rid="B29">29</xref>). There are 698 instances in the dataset with 9 categorical features (predictors). The features represented the characteristics of cell nuclei from breast masses sampled using fine-needle aspiration (FNA) (<xref ref-type="bibr" rid="B30">30</xref>). Possible values of each feature are 1 to 10, with 1 representing the closest to benign and 10 representing the closest to malignant. The outcome is a binary variable, which can either be benign or malignant. For simplicity, the dataset we used is relatively low dimensional, containing only 9 features, compared to most oncology research utilizing complex data with much more variables. Nevertheless, the interpretation approaches we discussed can be applied to models trained with high-dimensional data to provide rich insights beyond classification or regression outputs. Researchers have applied these methods to derive individualized, patient-centered information supporting clinical decision-making (<xref ref-type="bibr" rid="B31">31</xref>, <xref ref-type="bibr" rid="B32">32</xref>) and to uncover disease risk/protective factors from their prognostic or diagnostic models trained with complex, high-dimensional datasets (<xref ref-type="bibr" rid="B33">33</xref>, <xref ref-type="bibr" rid="B34">34</xref>).</p>
<p>We randomly split the dataset into a training set with 70% of the data for model development and a testing set with 30% of the data for model validation. For consistency and ease of reproducibility, we created our models with default configurations of the CARET (Classification And REgression Training) package without further hyperparameter optimization. We performed all modeling and analyses using the R statistical programming environment (version 4.2.1) (<xref ref-type="bibr" rid="B35">35</xref>) using CARET (version 6.0-93) (<xref ref-type="bibr" rid="B36">36</xref>), DALEX (version 2.4.2) (<xref ref-type="bibr" rid="B37">37</xref>), and lime (version 0.5.3) (<xref ref-type="bibr" rid="B38">38</xref>) packages. Reproducible code is available in the online supplement for all analyses.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Machine learning model interpretation approaches</title>
<p>Over the past decades since ML has been available, several approaches addressing interpretability issues of the ML-based models have been proposed and implemented (<xref ref-type="bibr" rid="B22">22</xref>). Some algorithms are interpretable-by-nature such as regularized logistic regression, nearest neighbors, and decision tree algorithms (<xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B39">39</xref>, <xref ref-type="bibr" rid="B40">40</xref>). We refer to these models as interpretable and refer to the interpretation methods as being model-specific. However, it can be practically impossible to comprehensively explain model outputs for models which rely on complete nonlinear data transformations, such as support vector machine and artificial neural networks, without applying model-agnostic approaches.</p>
<p>Moreover, model-specific approaches involve an understanding of the mechanism of the algorithm, whereas Linear Models, Decision Trees, etc. produce different explanations as results are highly impacted by feature selection and training hyperparameters. Interpretability in model-specific approaches is also undermined by feature complexity. Complex features (e.g., PCA-derived features), sparsity, lack of independence, monotonicity, and linearity do not guarantee interpretability. Finally, in some scenarios, particularly in larger datasets, simplicity may require sacrificing performance.</p>
<p>Model-agnostic approaches are a set of model interpretation methods that are applicable to ML models developed using any algorithms, including interpretable models. These methods can provide visualizations of model decision-making processes for human interpretation to answer questions, such as what the most important feature is for any model. The model-agnostic approaches can be further grouped into two categories, global and local interpretations. Global interpretations target uncovering average model decision-making processes at a dataset or cohort level, while local interpretations provide interpretations of model behaviors for individual predictions. Model agnostic approaches allow for flexibility in model choice, which means that there are more options to improve certain issues that may occur to a model in production that would necessitate the adoption of another algorithmic approach; a critical component to Safe and Effective ML Systems (<xref ref-type="bibr" rid="B13">13</xref>) and the FDA&#x2019;s Good Machine Learning Practices GMLP (<xref ref-type="bibr" rid="B41">41</xref>).</p>
<p>In the following section, we provide overview of each of the interpretation approaches along with an example showcasing the approach and its limitations. A summary of interpretation approaches covered in this paper is provided in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Summary of interpretation approaches covered. SHAP: Shapley additive explanation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g001.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Model-specific interpretation approaches</title>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Coefficient-based method</title>
<p>Model-specific approaches refer to model interpretation methods that are available as an inherent part of certain ML algorithms (<xref ref-type="bibr" rid="B26">26</xref>). One of the most widely known and accessible approaches for model interpretation is assessing coefficients that are available for many linear models. By investigating the coefficient of each feature included in the prediction, we can know which features were used by a model to make predictions and how each variable contributes to the model output. As an example, the coefficients of our GLM model for breast mass cancerous predictions are presented in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The coefficients indicate that all features were included in the model were positively associated with a prediction of malignancy.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Coefficients for the generalized linear model (GLM) with regularization.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">
Features
</th>
<th valign="top" align="center">
Coefficient
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">(Intercept)</td>
<td valign="top" align="center">-7.42</td>
</tr>
<tr>
<td valign="top" align="left">Normal Mitoses</td>
<td valign="top" align="center">0.70</td>
</tr>
<tr>
<td valign="top" align="left">Bland Chromatin</td>
<td valign="top" align="center">0.43</td>
</tr>
<tr>
<td valign="top" align="left">Adhesion</td>
<td valign="top" align="center">0.24</td>
</tr>
<tr>
<td valign="top" align="left">Cell Shape</td>
<td valign="top" align="center">0.23</td>
</tr>
<tr>
<td valign="top" align="left">Bare Nuclei</td>
<td valign="top" align="center">0.23</td>
</tr>
<tr>
<td valign="top" align="left">Cell Size</td>
<td valign="top" align="center">0.19</td>
</tr>
<tr>
<td valign="top" align="left">Epithelial Size</td>
<td valign="top" align="center">0.18</td>
</tr>
<tr>
<td valign="top" align="left">Normal Nucleoli</td>
<td valign="top" align="center">0.15</td>
</tr>
<tr>
<td valign="top" align="left">Thickness</td>
<td valign="top" align="center">0.06</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>Models adopting the MARS algorithm can be interpreted using a similar way. The MARS algorithm can be understood as an extension of a linear or rigid logistic regression model, which facilitates interactions between features whilst providing clear interpretability and deeper insight into the relationships in data (<xref ref-type="bibr" rid="B42">42</xref>, <xref ref-type="bibr" rid="B43">43</xref>). Thus, the same coefficient method for interpreting a GLM model can be applied to interpret a MARS model. <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> shows selected terms and their coefficients generated by our MARS model. We can use the information to calculate the probability of a breast mass sample being malignant and, in so doing, simulate the model behavior. The model identified interesting insights by revealing complex Thickness &#x2013; Cell Size, Cell Size &#x2013; Bare Nuclei, and Epithelial Size &#x2013; Bare Nuclei interactions, indicating various effects on model outputs depending on feature values. Clinical implications of the feature interactions identified may not be obvious in our example, but it becomes explicit and important if our model predicts Hemoglobin A1c (HbA1c) using age groups and Body Mass Index (BMI). Strengths of associations between BMI and HbA1c varies among different age groups, suggesting that different glycemic control strategies should be used for different age populations (<xref ref-type="bibr" rid="B44">44</xref>).</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Features and coefficients determined by the model using the multivariate adaptive regression splines (MARS) algorithm.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">No.</th>
<th valign="top" align="center">
Feature
</th>
<th valign="top" align="center">
Coefficient
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">1</td>
<td valign="top" align="center">(Intercept)</td>
<td valign="top" align="center">1.09</td>
</tr>
<tr>
<td valign="top" align="center">2</td>
<td valign="top" align="center">h(Cell_Size-2)</td>
<td valign="top" align="center">-0.32</td>
</tr>
<tr>
<td valign="top" align="center">3</td>
<td valign="top" align="center">h(2-Cell_Size)</td>
<td valign="top" align="center">-0.75</td>
</tr>
<tr>
<td valign="top" align="center">4</td>
<td valign="top" align="center">h(Cell_Size-3)</td>
<td valign="top" align="center">0.33</td>
</tr>
<tr>
<td valign="top" align="center">5</td>
<td valign="top" align="center">h(Bare_Nuclei-2)</td>
<td valign="top" align="center">-0.37</td>
</tr>
<tr>
<td valign="top" align="center">6</td>
<td valign="top" align="center">h(Bare_Nuclei-3)</td>
<td valign="top" align="center">0.42</td>
</tr>
<tr>
<td valign="top" align="center">7</td>
<td valign="top" align="center">h(Thickness-5)&#xd7;h(Cell_Size-2)</td>
<td valign="top" align="center">0.01</td>
</tr>
<tr>
<td valign="top" align="center">8</td>
<td valign="top" align="center">h(5-Thickness)&#xd7;h(Cell_Size-2)</td>
<td valign="top" align="center">0.02</td>
</tr>
<tr>
<td valign="top" align="center">9</td>
<td valign="top" align="center">h(Cell_Size-3)&#xd7;h(2-Bare_Nuclei)</td>
<td valign="top" align="center">0.99</td>
</tr>
<tr>
<td valign="top" align="center">10</td>
<td valign="top" align="center">h(3-Cell_Size)&#xd7;h(2-Bare_Nuclei)</td>
<td valign="top" align="center">-0.17</td>
</tr>
<tr>
<td valign="top" align="center">11</td>
<td valign="top" align="center">h(Cell_Size-2)&#xd7;h(2-Bare_Nuclei)</td>
<td valign="top" align="center">-0.83</td>
</tr>
<tr>
<td valign="top" align="center">12</td>
<td valign="top" align="center">h(2-Epithelial_Size)&#xd7;h(Bare_Nuclei-2)</td>
<td valign="top" align="center">0.21</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>h(Variable &#x2013; Constant) are hinge functions representing knots the multivariate adaptive regression splines model identified to better fit the data. The results of the functions are the maximum of 0 and the difference between the variable and constant values. For instance, suppose Cell_Size is 3, then h(Cell_Size-2)=Max(0, 3-2)=1.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Coefficients provide intuitive model interpretations to reveal model decision-making processes and enable easy implementation. Nevertheless, the approach provides less insight into the feature&#x2019;s effect at the individual level. Fixed coefficients revealed by a model may not reflect the variances in features&#x2019; effects on model outputs among individuals. Further, due to the use of regularization methods, there is a possibility that GLM and MARS models drop features that are clinically considered to be important and associated with outcomes the models predict (<xref ref-type="bibr" rid="B45">45</xref>).</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Rule-based decision tree method</title>
<p>Another widely recognized algorithm category is tree-based algorithms, which also allow intuitive interpretation whilst facilitating feature interactions (<xref ref-type="bibr" rid="B40">40</xref>). As their name suggests, these algorithms create a model by constructing a decision tree composed of a series of rules portioning data to determine model predictions. Although researchers have developed various tree-based algorithms, the method interpreting all models using these algorithms is the same if a single tree is developed (<xref ref-type="bibr" rid="B26">26</xref>). One can follow the rules of a tree-based model to reveal its decision-making process. We provided the decision tree and rule table used by our DT model as an example in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>. The tree is relatively small in our case, while one can get a huge tree containing hundreds of branches for a complex predicting task using a high-dimensional dataset. According to the rules, our DT model makes predictions using only Bare Nuclei and Cell Size variables. Our model classifies a breast mass as malignant only when the Bare Nuclei and Cell Size scores of the mass are greater than or equal to 2.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Decision tree (DT) and rule table.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g002.tif"/>
</fig>
<p>Although the interpretation allows an easy understanding of the behavior of a tree-based model, the method is limited to models based on a single tree. The method becomes less useful for the interpretation of models using many powerful tree-based algorithms, such as the random forest algorithm, due to the creation of multiple trees for making predictions. Although we can draw all the trees and go through each tree to understand how the models behave, it is impossible to know what key features are used by these models to drive decisions and how the features influence the decisions by using this approach.</p>
</sec>
<sec id="s3_1_3">
<label>3.1.3</label>
<title>Interpretation method for K-nearest neighbor models</title>
<p>A special class of ML models that allow interpretation without additional approaches are models using the <italic>k</italic>NN algorithm. A <italic>k</italic>NN model makes a prediction for a particular instance based on the neighbors of the instance (<xref ref-type="bibr" rid="B46">46</xref>). When predicting, a <italic>k</italic>NN model first identifies k instances most similar to the instance we are predicting from the training sample. Then, for a classification outcome, the model takes the most common class of the nearest neighbors identified. For continuous outcomes, the model averages the outcomes of the neighbors. Therefore, we can investigate the neighbors to understand the decision-making process of the model. For instance, we randomly selected a mass sample A from our validation sample and calculated the distance between features of A and all other masses in the training sample using the Euclidean distance method. As our KNN model used 7 nearest neighbors to determine predictions, the top seven instances with the smallest distances to A were the neighbors used by the model (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). As four out of the seven neighbors were benign, our <italic>k</italic>NN model predicted that A was not a cancerous mass. The Euclidean distance between the seven neighbors and the instance A were 4.5 &#xb1; 0.7, while the distance between all training data and the instance A were 10.2 &#xb1; 2.7.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Nearest neighbors of the example instance used by the k-Nearest Neighbor (KNN) model for prediction.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">Thickness</th>
<th valign="top" align="center">Cell size</th>
<th valign="top" align="center">Cell shape</th>
<th valign="top" align="center">Adhesion</th>
<th valign="top" align="center">Epithelial size</th>
<th valign="top" align="center">Bare nuclei</th>
<th valign="top" align="center">Bland Chromatin</th>
<th valign="top" align="center">Normal Nucleoli</th>
<th valign="top" align="center">Normal mitoses</th>
<th valign="top" align="center">Class</th>
<th valign="top" align="center">
<italic>d</italic>
</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Predicting instance (A)</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="top" align="left">Neighbor 1</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">malignant</td>
<td valign="top" align="center">3.5</td>
</tr>
<tr>
<td valign="top" align="left">Neighbor 2</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">benign</td>
<td valign="top" align="center">3.9</td>
</tr>
<tr>
<td valign="top" align="left">Neighbor 3</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">malignant</td>
<td valign="top" align="center">4.4</td>
</tr>
<tr>
<td valign="top" align="left">Neighbor 4</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">benign</td>
<td valign="top" align="center">4.5</td>
</tr>
<tr>
<td valign="top" align="left">Neighbor 5</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">benign</td>
<td valign="top" align="center">4.8</td>
</tr>
<tr>
<td valign="top" align="left">Neighbor 6</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">2</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">benign</td>
<td valign="top" align="center">5.4</td>
</tr>
<tr>
<td valign="top" align="left">Neighbor 7</td>
<td valign="top" align="center">10</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">5</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1</td>
<td valign="top" align="center">malignant</td>
<td valign="top" align="center">5.5</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>d: Euclidean distance to the predicting instance A.</p>
<p>- means Not Applicable.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Through observing the features of the nearest neighbors selected for predictions, we can uncover on what basis our models make predictions. This approach is like case-based rationale which we often adopt to make decisions in our daily life by looking at the cases/conditions similar to our current encounters from our past experiences. It also offers opportunities for examining whether the cohort of the current case was represented in the creation of the kNN model. However, the interpretation method delivers no information about whether a feature is weighted over other features. Further, the approach does not uncover whether a feature is positively or negatively associated with the outcome.</p>
<p>Beyond the interpretations of simple algorithms, researchers have spent substantial efforts to develop model-specific interpretation approaches for models using complex models that are generally considered as not interpretable-by-nature, such as random forest (<xref ref-type="bibr" rid="B47">47</xref>), support vector machine (<xref ref-type="bibr" rid="B48">48</xref>), and neural networks (<xref ref-type="bibr" rid="B49">49</xref>, <xref ref-type="bibr" rid="B50">50</xref>). Although these model-specific interpretation methods allow an easy understanding of model behaviors, the primary limitation is the limited flexibility of these methods. Use of these interpretation does not allow easy comparison among models using different algorithms Therefore, additional tools are needed to understand model decision-making processes when we pursue a higher-performing model with a more sophisticated algorithm.</p>
</sec>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Model-agnostic approaches</title>
<p>Model-agnostic interpretation approaches, contrary to model-specific methods, offer greater flexibility and can be applied to ML models using any algorithms. With flexibility, researchers can select any algorithms they believe are the best solution for solving the questions at hand and examine their models with consistent approaches for better model comparisons. These approaches use <italic>post hoc</italic> interpretation methods decomposing trained ML models (<xref ref-type="bibr" rid="B22">22</xref>). The general idea of the approach is to reveal model behaviors by observing the changes in model predictions when manipulating the input data instead of breaking down the models for an understanding of model structures. In the following section, we provided a gentle introduction to a few widely used approaches covering both global and local model-agnostic interpretation methods. Enthusiastic readers can find introductions to other model-agonistic methods in Molnar&#x2019;s book addressing interpretability issues of ML (<xref ref-type="bibr" rid="B26">26</xref>).</p>
<sec id="s3_2_1">
<label>3.2.1</label>
<title>Global interpretation</title>
<p>The global interpretation methods focus on providing an overall picture depicting model behaviors at a dataset level. The approaches can help to reveal the averaged effect of a feature on model predictions for a given dataset. Of the global interpretation methods developed, feature importance (FI), partial dependence plots (PDP), and accumulated local effects (ALE) were the most widely used approaches in literature. We provide a description of these methods and demonstrate their utilization with our breast cancerous prediction model using the extreme gradient boosting trees algorithm (XGBT model).</p>
<sec id="s3_2_1_1">
<label>3.2.1.1</label>
<title>Feature importance</title>
<p>A frequent question for a given predictive model beyond model performance is what features are important to the model for accurate predictions, which can be addressed by the FI analysis (<xref ref-type="bibr" rid="B51">51</xref>). The FI analysis estimates the importance of a feature by calculating model performance changes (e.g., loss in area under the receiver operating characteristic curve) when we randomly alter the feature&#x2019;s value (<xref ref-type="bibr" rid="B52">52</xref>). A feature is deemed important if the performance loss is notable when permutating the feature&#x2019;s value. Taking our XGBT model as an example, the FI analysis shows that cell shape, bare nuclei, normal mitoses, and epithelial size scores were the most important features enabling the model to generate accurate outputs (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>).</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Feature importance (FI) analysis for the extreme gradient boosting tree model. AUC, Area under the receiver-operating characteristics.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g003.tif"/>
</fig>
<p>The FI analysis is widely recognized as a useful approach allowing provision of compressed insights into model behaviors and is commonly utilized in medical ML literature (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B53">53</xref>, <xref ref-type="bibr" rid="B54">54</xref>). However, we should note that the result of the analysis does not reveal how features affect model decisions (<xref ref-type="bibr" rid="B52">52</xref>, <xref ref-type="bibr" rid="B53">53</xref>). For instance, the FI result delivered no information on whether our XGBT model assigns a greater cancerous probability to a sample with a higher value in bland chromatin. In addition, due to the use of random permutation and intrinsic machine-selection of features, correlations between features can be problematic and result in unreliable feature importance estimates.</p>
</sec>
<sec id="s3_2_1_2">
<label>3.2.1.2</label>
<title>Partial dependence plot</title>
<p>In addition to important features, we may also be interested in knowing how the values of important features affect model predictions. A popular approach to address this question is to use partial dependence plots (PDP) to visualize the relationship between the outcome and a predicting feature of interest (<xref ref-type="bibr" rid="B22">22</xref>). The idea of the method is to estimate the relationships by marginalizing the feature of interest and calculating its marginal distribution (<xref ref-type="bibr" rid="B55">55</xref>, <xref ref-type="bibr" rid="B56">56</xref>). For instance, suppose we want to know the relationship between the bare nuclei score of a breast mass sample and the predictions generated by our XGBT model; we can fix the value of the feature for all instances in the validation dataset and calculate a mean predicted malignancy probability. Next, we calculate the mean predicted probabilities for all possible values of the bare nuclei score (<xref ref-type="bibr" rid="B1">1</xref>&#x2013;<xref ref-type="bibr" rid="B10">10</xref>) to uncover the probability distribution (marginal distribution). Through plotting out the probability distribution, we observe a positive relationship between the feature and the predicted malignancy probability generated by the model (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Relationship between bare nuclei score and predicted cancerous probability generated by our extreme gradient boosting tree (XGBT) model using partial dependence analysis.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g004.tif"/>
</fig>
<p>Although PDPs provide useful and intuitive model behavior interpretation, there are disadvantages of the approach that are important to highlight. First, the method assumes no interaction between features, which is not likely to be the truth for a real-world clinical dataset (<xref ref-type="bibr" rid="B26">26</xref>, <xref ref-type="bibr" rid="B57">57</xref>). The approach can estimate feature effects based on unrealistic data. It is not apparent with our breast mass cancerous example. However, if our model was to predict house prices using room numbers and surface space, the approach could generate unrealistic data, such as ten rooms within a 100 square feet house. In such cases, the approach is not useful and can generate misleading results. Another limitation of PDPs was the use of a mean predicted probability and disregarding the distribution of the predicted probabilities when estimating the probabilities of the interested feature fixed with a certain value. The PDP results become less meaningful or even misleading if the distribution is scattered (<xref ref-type="bibr" rid="B26">26</xref>).</p>
</sec>
<sec id="s3_2_1_3">
<label>3.2.1.3</label>
<title>Accumulated local effect</title>
<p>Another popular global model interpretation method is ALE, which addresses the same question as PDP does, while ALE provides more reliable model behavior information when correlations between features exist (<xref ref-type="bibr" rid="B57">57</xref>). The primary difference between the approaches is that ALE performs marginalization locally within instances with a similar value of the feature we are examining to avoid the use of unrealistic data for estimating model behaviors. Further, ALE uses differences in predicted probabilities generated for instances with similar values for the feature of interest as an alternative of mean to avoid the issue of a scattered distribution (<xref ref-type="bibr" rid="B57">57</xref>).</p>
<p>We again examined the effect of the bare nuclei score on the outputs of our XGBT model but using ALE. The ALE graph indicated the predicted probabilities notably increased when the bare nuclei score was ten, while the probabilities reduced when the feature was scored two (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>). All other possible values of the feature resulted in similar predictions. The result is remarkably different from the PDP result. As correlations were likely to exist among the features in the breast mass dataset, we argued that the ALE provides more reliable information regarding the impact of specific features on model behavior.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Accumulated local effect (ALE) analysis.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g005.tif"/>
</fig>
<p>The approach is not without limitations. The results of ALE are more complex compared to PDP and less interpretable, especially when strong and complex correlations between features exist (<xref ref-type="bibr" rid="B58">58</xref>). The ALE can still generate unstable interpretations of feature effects due to the arbitrary selection of numbers of intervals where local feature effects are estimated (<xref ref-type="bibr" rid="B11">11</xref>). Further, as ALE estimates feature effects per interval, the interpretation is interval specific and may not be applicable to other intervals (<xref ref-type="bibr" rid="B26">26</xref>). Nevertheless, the approach provides visual, unbiased interpretation of feature effects on model predictions and is recommended for interpreting models trained with clinical data that often involve correlated features (<xref ref-type="bibr" rid="B26">26</xref>).</p>
</sec>
</sec>
<sec id="s3_2_2">
<label>3.2.2</label>
<title>Local interpretation</title>
<p>Thus far, we have introduced several methods to uncover general model behaviors at the dataset level. As the primary utilizations of ML models are to provide individualized predictions, we may be interested in how a model makes predictions for individuals based on their data. Local interpretation methods were developed to uncover how much the value of each feature of an individual contributes to the ML model output for the individual to provide additional insights enabling individualized care (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B27">27</xref>). In this section, we cover commonly used local interpretation approaches following the same structure we used in the previous section for global interpretations. To demonstrate the methods, we randomly selected an instance from the validation sample and examined the prediction generated by our NNET model for this instance. We provide the characteristics of the mass sample selected in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Feature of the mass sample selected.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Feature</th>
<th valign="top" align="center">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Thickness</td>
<td valign="top" align="center">8</td>
</tr>
<tr>
<td valign="top" align="center">Cell size</td>
<td valign="top" align="center">4</td>
</tr>
<tr>
<td valign="top" align="center">Cell shape</td>
<td valign="top" align="center">5</td>
</tr>
<tr>
<td valign="top" align="center">Adhesion</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td valign="top" align="center">Epithelial size</td>
<td valign="top" align="center">2</td>
</tr>
<tr>
<td valign="top" align="center">Bare nuclei</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td valign="top" align="center">Bland Chromatin</td>
<td valign="top" align="center">7</td>
</tr>
<tr>
<td valign="top" align="center">Normal Nucleoli</td>
<td valign="top" align="center">3</td>
</tr>
<tr>
<td valign="top" align="center">Normal mitoses</td>
<td valign="top" align="center">1</td>
</tr>
<tr>
<td valign="top" align="center">Class</td>
<td valign="top" align="center">Malignant</td>
</tr>
</tbody>
</table>
</table-wrap>
<sec id="s3_2_2_1">
<label>3.2.2.1</label>
<title>Break down plot</title>
<p>One of the most straightforward approaches to examine the feature contributions to individual predictions is using a Break Down (BP) plot. The approach decomposes a model prediction into contributions and it then estimates the attribution of the contributions from each feature (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B58">58</xref>). The intuition of the interpretation is to estimate the mean predictions for each feature when we consecutively fix an exploratory feature and permutate all other features (<xref ref-type="bibr" rid="B59">59</xref>). For instance, to examine the attribution for the sample we selected, we first computed the mean prediction by fixing the bare nuclei score to 1 and permutating all other features. As shown in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6A</bold>
</xref>, we got a mean prediction lower than the prediction for an intercept model by 0.02, indicating that having a bare nuclei score of 1 lowers the cancerous probability for the selected sample. For the next feature, we fix both the bare nuclei and cell shape score to calculate the mean prediction change. The process continued until all feature values were fixed.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Break down (BD) plots showing how the contributions attributed to individual features for the instance we selected. <bold>(A)</bold> break down plot assuming no interactions among features; <bold>(B)</bold> break down plot with feature interactions considered.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g006.tif"/>
</fig>
<p>Break down plots provide clear visualizations for us to evaluate feature contributions to individual predictions made by ML models. However, the disadvantage of the approach is that the order in which we examine features can significantly alters the result. The approach can provide misleading interpretations if feature interactions exist and the order is not carefully determined (<xref ref-type="bibr" rid="B58">58</xref>). When interactions between features exist, the interaction version of BD plot should be considered and may provide better information to address the ordering issue (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6B</bold>
</xref>). However, the BD plots for interactions can be computationally expensive and hard to understand for large feature numbers (<xref ref-type="bibr" rid="B58">58</xref>).</p>
</sec>
<sec id="s3_2_2_2">
<label>3.2.2.2</label>
<title>Local surrogate</title>
<p>Another approach to decompose individual predictions is to create an interpretable model (such as a linear or decision tree model) as a surrogate of our model and approximate model behavior by investigating the surrogate model (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B22">22</xref>). We direct interested readers to the original paper for a detailed description of creating surrogate models (<xref ref-type="bibr" rid="B8">8</xref>). <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> shows the result of using a linear surrogate model to reveal why our NNET model assigns the malignant class to the breast mass sample we selected. Since it is a linear surrogate model, we can visualize feature effects using the coefficients determined by the surrogate model for the sample. For instance, the surrogate model estimated that the bland chromatin score of this sample increased the predicted likelihood of being a malignant tumor by 0.26. This approach focuses on decomposing individual predictions, and thus the surrogate model can be used to investigate feature effects for the mass sample we used to create the model. For other samples, we will need to create other surrogate models using their own data for interpretation.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Feature effects on neural network (NNET) model prediction for the breast mass sample randomly selected using a local surrogate model approach.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g007.tif"/>
</fig>
<p>In addition to the tabular data we have shown, the approach is also useful in interpreting models using text and image data, allowing easy interpretation for models using any data types and algorithms (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B26">26</xref>). Nevertheless, the approach has several unsolved issues in surrogate model creation processes, such as the methods adopted to select training data and determine the weights of each training data point. The results generated by surrogate models created may vary for the same individual prediction due to the use of different data perturbation, feature selection and weighting methods (<xref ref-type="bibr" rid="B11">11</xref>, <xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B26">26</xref>).</p>
</sec>
<sec id="s3_2_2_3">
<label>3.2.2.3</label>
<title>SHapely additive exPlanations</title>
<p>The SHapely additive exPlanations (SHAP) approach is another tool that provides local interpretation to drive additional insights into feature effects on individual predictions of black-box ML models. The approach used Shapley values from cooperative game theory addressing how to fairly distribute contributions to players cooperatively finishing a game. In the scenario of ML, features are players, and contributions are the differences in model predictions between the instance of interest and other instances with similar characteristics. Thus, SHAP values are useful in approximating feature contributions to individual predictions of a black-box model. The intuition of the approach is detailed in the original publications (<xref ref-type="bibr" rid="B9">9</xref>). In short, the approach used a permutation process similar to the break down plots, while the SHAP approach takes mean probability differences across many or all possible orderings as outputs to avoid the ordering issue (<xref ref-type="bibr" rid="B58">58</xref>). Using the SHAP approach, we examined the feature effects on the NNET model prediction for our breast mass example and revealed that the bland chromatin score and thickness score for the sample is the most salient positive and negative contributors, respectively (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>).</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Insights into feature effects on model predictions from the Shapley additive explanation (SHAP) analysis.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g008.tif"/>
</fig>
<p>The approach has gained popularity in the past few years and is suggested for deriving additional insights into feature effects in health literature using ML (<xref ref-type="bibr" rid="B32">32</xref>, <xref ref-type="bibr" rid="B60">60</xref>). The major limitations of the approach include computational expense for a large model with many features, the requirement of the training dataset to enable the permutation process, and the inclusion of unrealistic data during the premutation process (<xref ref-type="bibr" rid="B58">58</xref>).</p>
</sec>
<sec id="s3_2_2_4">
<label>3.2.2.4</label>
<title>Ceteris-Paribus plot</title>
<p>The last local interpretation approach we covered is Ceteris-Paribus (CP) plots, also named individual conditional expectations (ICE), that address &#x201c;what-if&#x201d; questions to provide insights into individual model predictions (<xref ref-type="bibr" rid="B61">61</xref>, <xref ref-type="bibr" rid="B62">62</xref>). The approach evaluates the effect of a feature on model predictions by calculating prediction changes when replacing the value of the feature with values of all other features fixed (<xref ref-type="bibr" rid="B58">58</xref>). For instance, if we want to examine the dependence between cell shape scores and the NNET model output for the breast mass we selected, we can have the model make predictions on a set of samples with each having a possible score for cell shape and other features with the same value of the selected breast mass sample. Then, we can visualize the predictions to investigate how changes in cell shape scores influence model outputs (<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>).</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Conditional dependence between cell shape and the neural network prediction for our random-selected breast mass sample. The graph indicates that the cancerous probabilities increase if the sample&#x2019;s cell shape score increases. The blue dot represents the observed score of cell shape for the sample we selected and the corresponding model prediction.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-13-1129380-g009.tif"/>
</fig>
<p>The CP plots provide a counterfactual interpretation to quantify feature effects and offer clear visualization to investigate the relationships between model responses and features (<xref ref-type="bibr" rid="B62">62</xref>). However, the approach is limited to displaying information for one feature at a time. When the feature number is large, using the approach to decompose model predictions becomes overwhelming because many plots need to be drawn and interpreted. In addition, the approach also assumes no interactions among features (<xref ref-type="bibr" rid="B58">58</xref>). Therefore, unrealistic data could be included by the approach to provide misleading information when feature interactions exist.</p>
</sec>
</sec>
</sec>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>The black-box model consideration remains one of the biggest challenges to clinical implementation of ML-based tools to inform clinical decisions for oncology care (<xref ref-type="bibr" rid="B60">60</xref>, <xref ref-type="bibr" rid="B63">63</xref>&#x2013;<xref ref-type="bibr" rid="B65">65</xref>). As a fast-emerging field, researchers have developed many interpretation approaches deconstructing model predictions from varying aspects to provide additional insights into model predictions. In this manuscript, we provide introductions to various model interpretation techniques, including model-specific, model-agnostic global, and model-agnostic local interpretations, with accompanying examples showing the information these approaches offer along with their respective advantage and disadvantages. Each interpretation provides different insights regarding model behaviors and the effects of the input features. We suggest using these techniques to provide additional insights beyond simple model outputs will help future ML studies in the oncology field translate to the clinic through improved interpretability.</p>
<p>Oncology patients are vulnerable and require carefully planned treatments. Oncologists are often more reluctant to take suggestions without explanations on how the suggestions were generated, resulting in low adoption of the ML-based decision support tools in the field (<xref ref-type="bibr" rid="B66">66</xref>, <xref ref-type="bibr" rid="B67">67</xref>). Use of the model interpretations enables information concerning model decision-making processes beyond model outputs. Providing oncologists with this information accompanying model suggestions may be the key to increasing their adoption and enabling the full potential capability of ML models to enhance oncology care (<xref ref-type="bibr" rid="B32">32</xref>). Although future research is needed to reveal the impacts of such information on patient care and outcomes, we encourage the use of model interpretation approaches in research and implementation work to examine model decisions, explore their impacts on model development and care practices, and drive novel insights from data into future research.</p>
<p>Various model interpretation techniques are available, and each has its own advantages, disadvantage, and use cases. For instance, model-specific interpretations provide intuitive interpretations by revealing actual model structure, while the utilization of the approaches is limited to models using specific ML algorithms (<xref ref-type="bibr" rid="B22">22</xref>, <xref ref-type="bibr" rid="B26">26</xref>). On the other hand, model-agnostic approaches can be applied to any ML models, including ensemble models using multiple ML algorithms, to facilitate the decomposition of varying ML models in the same way to enable comparison between models (<xref ref-type="bibr" rid="B26">26</xref>). However, appropriate selection of the approaches to use can be challenging and depends on the characteristics of the datasets used for model training and validation. Use of inappropriate approaches, such as applying PDP to a dataset containing intercorrelated features, can generate misleading information that is not easy to distinguish and may result in unintentional harm (<xref ref-type="bibr" rid="B68">68</xref>). Unfortunately, there is no guideline or standard guiding the use of these approaches, however, increasing the awareness of these techniques in the oncology community is an important initial step to establishing the interdisciplinary collaboration involving clinical experts, data scientists, and ML engineers that will lead to more robust interpretation.</p>
<p>Model interpretations, including both model-specific and -agnostic approaches, offer additional benefits beyond uncovering model behaviors by allowing us opportunities to detect biased data and quality issues in our data for model improvement (<xref ref-type="bibr" rid="B63">63</xref>). For instance, the interpretations can detect a feature value leading to a certain model decision that is contradictory to clinical knowledge, indicating potential data issues. In a previous analysis using SHAP to explore decision-making processes of models by our team, we showed that alcohol use could protect patients using immune checkpoint inhibitors from short-term readmission [manuscript in press]. This could be a manifestation of reporting bias and data granularity issues instead of related to alcohol consumption. People can be self-selecting in reporting their drinking status and do not always disclose alcohol use, especially heavy use. There could be different levels of use among alcohol users. Light alcohol use may have benefits, while heavy use is obviously harmful. In our case, we used a binary feature to represent alcohol use status that might not be enough to reveal the true effects of alcohol consumption and reduce the discrimination of our models.</p>
<p>Model-agnostic interpretations are capable of enabling additional insights into data health assessment as they are fully data-driven approaches. Data drifting, defined as variations in data used for model development from the data used for model validation and enabling the model after deployment, is a concept that has been increasingly discussed in the ML literature (<xref ref-type="bibr" rid="B69">69</xref>, <xref ref-type="bibr" rid="B70">70</xref>). A key factor leading to data variations is time. The meaning, measurement, or definition of a feature enabling the functioning of a model can change over time and result in degradation in model performance or even outright malfunction. For instance, the definition of a disease can change in a short period of time with new evidence discovered. This is particularly true as new markers and therapeutics emerge in the healthcare industry, especially for the oncology area (<xref ref-type="bibr" rid="B70">70</xref>). A model may become irrelevant whenever data drifts and continue to provide outputs without any realization of the change in inputs. Use of model-agnostic approaches allows us to detect model dysfunctions by revealing a significant change in model decision-making processes before and after data drift (<xref ref-type="bibr" rid="B71">71</xref>).</p>
<p>Despite many advantages, a few general limitations exist across the interpretation approaches in addition to the disadvantages discussed in the previous sections. Model interpretations are not detached from model performance. Misleading information can be a result of interpreting under- or over-fitted models (<xref ref-type="bibr" rid="B63">63</xref>, <xref ref-type="bibr" rid="B68">68</xref>). Therefore, we suggest prioritizing model generalizability and applying the interpretation approaches to those high-performing models for additional insights. For model-agnostic approaches, they are incapable of depicting models&#x2019; underlying mechanisms of how they process input data to generate decisions. An argument is that the approaches only uncover certain aspects of models that are human-intelligible and leave other parts still in a black box (<xref ref-type="bibr" rid="B63">63</xref>). Further, most model-agnostic approaches provide no information on their fidelity to the original models and do not quantify uncertainty generated during the resampling and perturbation (<xref ref-type="bibr" rid="B51">51</xref>, <xref ref-type="bibr" rid="B63">63</xref>, <xref ref-type="bibr" rid="B68">68</xref>).</p>
<p>Although there is growing awareness of the need, research in interpretability ML is still in its infant stage and requires more attention. Misuse of the interpretation approaches is likely and can result in unpleasant consequences (<xref ref-type="bibr" rid="B68">68</xref>, <xref ref-type="bibr" rid="B72">72</xref>). One future effort can be the development of guidelines for researchers to select approaches suitable to their models and data. Moreover, to our knowledge, these approaches were used mainly in model development and in-silo validation, and less attention was on the impacts of this additional information on care practices and patient outcomes. Increasing in awareness of model interpretability is an essential first step to enabling interdisciplinary approaches for the development and implementation of robust, interpretable ML models (<xref ref-type="bibr" rid="B11">11</xref>). Future prospective studies may be feasible to enable thorough suggestions concerning applications and utilizations of the approaches.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions</title>
<p>Many ML applications have been developed to support oncology care, but the adoption of the tools among oncologists is low due to challenges in model performance and reproducibility across settings. Introducing interpretability of models can inform poor performance and data quality issues that in turn can be helpful in model development and implementation. In this paper, we provide an accessible introduction to the ideas, use cases, advantages, and limitations of several commonly used model interpretation approaches. We encourage the use of various model-agnostic approaches in ML work supporting oncology care to derive enriched insights from clinical data and report models alongside additional model decision-making process information to allow model utilization and adoption appraisal. Further investigations on the impacts and communication of the model interpretations are needed to enable better utilization of the approaches.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>Conception and design: CS-G, S-CL, Collection and assembly of data: S-CL; Data analysis and interpretation: CS-G, S-CL, CS, CC, DJ. All authors contributed to the article and approved the submitted version.fonc.2023.1129380</p>
</sec>
</body>
<back>
<ack>
<title>Acknowledgments</title>
<p>We acknowledge the support to this work from the Institute for Data Science in Oncology at the University of Texas MD Anderson Cancer Center.</p>
</ack>
<sec id="s7" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s8" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nardini</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Machine learning in oncology: A review</article-title>. <source>Ecancermedicalscience</source> (<year>2020</year>) <volume>14</volume>:<elocation-id>1065</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3332/ECANCER.2020.1065</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Melstrom</surname> <given-names>LG</given-names>
</name>
<name>
<surname>Rodin</surname> <given-names>AS</given-names>
</name>
<name>
<surname>Rossi</surname> <given-names>LA</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>P</given-names>
</name>
<name>
<surname>Fong</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>V</given-names>
</name>
</person-group>. <article-title>Patient generated health data and electronic health record integration in oncologic surgery: A call for artificial intelligence and machine learning</article-title>. <source>J Surg Oncol</source> (<year>2021</year>) <volume>123</volume>:<fpage>52</fpage>&#x2013;<lpage>60</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/JSO.26232</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dlamini</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Francies</surname> <given-names>FZ</given-names>
</name>
<name>
<surname>Hull</surname> <given-names>R</given-names>
</name>
<name>
<surname>Marima</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Artificial intelligence (AI) and big data in cancer and precision oncology</article-title>. <source>Comput Struct Biotechnol J</source> (<year>2020</year>) <volume>18</volume>:<page-range>2300&#x2013;11</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.csbj.2020.08.019</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramesh</surname> <given-names>S</given-names>
</name>
<name>
<surname>Chokkara</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Major</surname> <given-names>A</given-names>
</name>
<name>
<surname>Volchenboum</surname> <given-names>SL</given-names>
</name>
<name>
<surname>Mayampurath</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Applications of artificial intelligence in pediatric oncology: A systematic review</article-title>. <source>JCO Clin Cancer Inform</source> (<year>2021</year>) <volume>5</volume>:<page-range>1208&#x2013;19</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/CCI.21.00102</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Balachandran</surname> <given-names>VP</given-names>
</name>
<name>
<surname>Gonen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>DeMatteo</surname> <given-names>RP</given-names>
</name>
</person-group>. <article-title>Nomograms in oncology: More than meets the eye</article-title>. <source>Lancet Oncol</source> (<year>2015</year>) <volume>16</volume>:<page-range>e173&#x2013;80</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S1470-2045(14)71116-7</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pfob</surname> <given-names>A</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>SC</given-names>
</name>
<name>
<surname>Sidey-Gibbons</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Machine learning in medicine: A practical introduction to techniques for data pre-processing, hyperparameter tuning, and model comparison</article-title>. <source>BMC Med Res Methodol</source> (<year>2022</year>) <volume>22</volume>:<fpage>1</fpage>&#x2013;<lpage>15</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12874-022-01758-8</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sidey-Gibbons</surname> <given-names>JAM</given-names>
</name>
<name>
<surname>Sidey-Gibbons</surname> <given-names>CJ</given-names>
</name>
</person-group>. <article-title>Machine learning in medicine: A practical introduction</article-title>. <source>BMC Med Res Methodol</source> (<year>2019</year>) <volume>19</volume>:<fpage>1</fpage>&#x2013;<lpage>18</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12874-019-0681-4</pub-id>
</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ribeiro</surname> <given-names>MT</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>S</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C</given-names>
</name>
</person-group>. (<year>2016</year>). <article-title>Why should I trust you</article-title>? Explaining the predictions of any classifier, in: <conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>, <conf-loc>New York, NY, USA</conf-loc> (<publisher-loc>San Francisco, CA</publisher-loc>: <publisher-name>Association for Computing Machinery (ACM)</publisher-name>). pp. <page-range>1135&#x2013;44</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/2939672.2939778</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>SI</given-names>
</name>
</person-group>. <article-title>A unified approach to interpreting model predictions</article-title>. In: <source>Advances in neural information processing systems</source>. <publisher-loc>CA, USA</publisher-loc>: <publisher-name>Long Beach</publisher-name> (<year>2017</year>). p. <page-range>4766&#x2013;75</page-range>.</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adadi</surname> <given-names>A</given-names>
</name>
<name>
<surname>Berrada</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Peeking inside the black-box: A survey on explainable artificial intelligence (XAI)</article-title>. <source>IEEE Access</source> (<year>2018</year>) <volume>6</volume>:<page-range>52138&#x2013;60</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2018.2870052</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carvalho</surname> <given-names>DV</given-names>
</name>
<name>
<surname>Pereira</surname> <given-names>EM</given-names>
</name>
<name>
<surname>Cardoso</surname> <given-names>JS</given-names>
</name>
</person-group>. <article-title>Machine learning interpretability: A survey on methods and metrics</article-title>. <source>Electron (Basel)</source> (<year>2019</year>) <volume>8</volume>:<elocation-id>832</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/electronics8080832</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Gilpin</surname> <given-names>LH</given-names>
</name>
<name>
<surname>Bau</surname> <given-names>D</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>BZ</given-names>
</name>
<name>
<surname>Bajwa</surname> <given-names>A</given-names>
</name>
<name>
<surname>Specter</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kagal</surname> <given-names>L</given-names>
</name>
</person-group>. (<year>2019</year>). <article-title>Explaining explanations: An overview of interpretability of machine learning</article-title>, in: <conf-name>Proceedings - 2018 IEEE 5th International Conference on Data Science and Advanced Analytics, DSAA 2018</conf-name>, (<publisher-loc>Turin, Italy</publisher-loc>: <publisher-name>IEEE</publisher-name>). pp. <page-range>80&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/DSAA.2018.00018</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="web">
<source>White house office of science and technology policy. blueprint for an AI bill of rights</source> (<year>2022</year>). Available at: <uri xlink:href="https://www.whitehouse.gov/wp-content/uploads/2022/10/Blueprint-for-an-AI-Bill-of-Rights.pdf">https://www.whitehouse.gov/wp-content/uploads/2022/10/Blueprint-for-an-AI-Bill-of-Rights.pdf</uri> (Accessed <access-date>December 8, 2022</access-date>).</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="web">
<source>U.S. department of health and human services food and drug administration. clinical decision support software - guidance for industry and food and drug administration staff</source> . Available at: <uri xlink:href="https://www.fda.gov/drugs/guidance-">https://www.fda.gov/drugs/guidance-</uri> (Accessed <access-date>December 8, 2022</access-date>).</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="book">
<person-group person-group-type="author">
<collab>IT Governance Privacy Team</collab>
</person-group>. <article-title>EU General data protection regulation (GDPR)</article-title>. In: <source>An implementation and compliance guide</source>, <edition>3rd ed</edition>. <publisher-loc>Itgp, United Kingdom</publisher-loc>: <publisher-name>IT Governance Publishing</publisher-name> (<year>2019</year>). doi:&#xa0;<pub-id pub-id-type="doi">10.2307/J.CTVR7FCWB</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bertsimas</surname> <given-names>D</given-names>
</name>
<name>
<surname>Wiberg</surname> <given-names>H</given-names>
</name>
</person-group>. <article-title>Machine learning in oncology: Methods, applications, and challenges</article-title>. <source>JCO Clin Cancer Inform</source> (<year>2020</year>) <volume>4</volume>:<page-range>885&#x2013;94</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/cci.20.00072</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cuocolo</surname> <given-names>R</given-names>
</name>
<name>
<surname>Caruso</surname> <given-names>M</given-names>
</name>
<name>
<surname>Perillo</surname> <given-names>T</given-names>
</name>
<name>
<surname>Ugga</surname> <given-names>L</given-names>
</name>
<name>
<surname>Petretta</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Machine learning in oncology: A clinical appraisal</article-title>. <source>Cancer Lett</source> (<year>2020</year>) <volume>481</volume>:<fpage>55</fpage>&#x2013;<lpage>62</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.canlet.2020.03.032</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parikh</surname> <given-names>RB</given-names>
</name>
<name>
<surname>Hasler</surname> <given-names>JS</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>M</given-names>
</name>
<name>
<surname>Chivers</surname> <given-names>C</given-names>
</name>
<name>
<surname>Ferrell</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>Development of machine learning algorithms incorporating electronic health record data, patient-reported outcomes, or both to predict mortality for outpatients with cancer</article-title>. <source>JCO Clin Cancer Inform</source> (<year>2022</year>) <volume>6</volume>:<elocation-id>e2200073</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/CCI.22.00073</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname> <given-names>S-C</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>C</given-names>
</name>
<name>
<surname>Nguyen</surname> <given-names>CH</given-names>
</name>
<name>
<surname>Geng</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Pfob</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sidey-Gibbons</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Machine learning-based short-term mortality prediction models for cancer patients using electronic health record data: A systematic review and critical appraisal</article-title>. <source>JMIR Med Inform</source> (<year>2022</year>) <volume>10</volume>:<elocation-id>e33182</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.2196/33182</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nagy</surname> <given-names>M</given-names>
</name>
<name>
<surname>Radakovich</surname> <given-names>N</given-names>
</name>
<name>
<surname>Nazha</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Machine learning in oncology: What should clinicians know</article-title>? <source>JCO Clin Cancer Inform</source> (<year>2020</year>) <volume>4</volume>, <fpage>799</fpage>&#x2013;<lpage>810</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/cci.20.00049</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoon</surname> <given-names>CH</given-names>
</name>
<name>
<surname>Torrance</surname> <given-names>R</given-names>
</name>
<name>
<surname>Scheinerman</surname> <given-names>N</given-names>
</name>
</person-group>. <article-title>Machine learning in medicine: Should the pursuit of enhanced interpretability be abandoned</article-title>? <source>J Med Ethics</source> (<year>2022</year>) <volume>48</volume>:<page-range>581&#x2013;5</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/medethics-2020-107102</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Linardatos</surname> <given-names>P</given-names>
</name>
<name>
<surname>Papastefanopoulos</surname> <given-names>V</given-names>
</name>
<name>
<surname>Kotsiantis</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Explainable AI: A review of machine learning interpretability methods</article-title>. <source>Entropy</source> (<year>2020</year>) <volume>23</volume>:<elocation-id>18</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/e23010018</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Madsen</surname> <given-names>A</given-names>
</name>
<name>
<surname>Reddy</surname> <given-names>S</given-names>
</name>
<name>
<surname>Chandar</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Post-hoc interpretability for neural NLP: A survey</article-title>. <source>ACM Comput Surv</source> (<year>2021</year>) <volume>55</volume>(<issue>8</issue>):<page-range>1&#x2013;42</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arxiv.2108.04840</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yuan</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>T</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>C</given-names>
</name>
<name>
<surname>Du</surname> <given-names>M</given-names>
</name>
<name>
<surname>Johnson</surname> <given-names>BE</given-names>
</name>
<name>
<surname>Lanuti</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Performance of a machine learning algorithm using electronic health record data to identify and estimate survival in a longitudinal cohort of patients with lung cancer</article-title>. <source>JAMA Netw Open</source> (<year>2021</year>) <volume>4</volume>:<fpage>e2114723</fpage>&#x2013;<lpage>e2114723</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/jamanetworkopen.2021.14723</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bibault</surname> <given-names>JE</given-names>
</name>
<name>
<surname>Giraud</surname> <given-names>P</given-names>
</name>
<name>
<surname>Burgun</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Big data and machine learning in radiation oncology: State of the art and future prospects</article-title>. <source>Cancer Lett</source> (<year>2016</year>) <volume>382</volume>:<page-range>110&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.CANLET.2016.05.033</pub-id>
</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Molnar</surname> <given-names>C</given-names>
</name>
</person-group>. <source>Interpretable machine learning: A guide for making black box models explainable</source>. <edition>2nd</edition>. <publisher-name>Independently published</publisher-name> (<year>2022</year>). Available at: <uri xlink:href="https://christophm.github.io/interpretable-ml-book/">https://christophm.github.io/interpretable-ml-book/</uri>.</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Murdoch</surname> <given-names>WJ</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>C</given-names>
</name>
<name>
<surname>Kumbier</surname> <given-names>K</given-names>
</name>
<name>
<surname>Abbasi-Asl</surname> <given-names>R</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Definitions, methods, and applications in interpretable machine learning</article-title>. <source>Proc Natl Acad Sci U.S.A.</source> (<year>2019</year>) <volume>116</volume>:<page-range>22071&#x2013;80</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1073/pnas.1900654116</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hakkoum</surname> <given-names>H</given-names>
</name>
<name>
<surname>Abnane</surname> <given-names>I</given-names>
</name>
<name>
<surname>Idri</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Interpretability in the medical field: A systematic mapping and review study</article-title>. <source>Appl Soft Comput</source> (<year>2022</year>) <volume>117</volume>:<elocation-id>108391</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.ASOC.2021.108391</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Dua</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Graff</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2019</year>). <source>UCI machine learning repository: Mammographic mass data set</source> (<publisher-loc>Irvine, CA</publisher-loc>: <publisher-name>University of California, School of Information and Computer Science</publisher-name>). <uri xlink:href="http://archive.ics.uci.edu/ml">http://archive.ics.uci.edu/ml</uri>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ferris</surname> <given-names>MC</given-names>
</name>
<name>
<surname>Mangasarian</surname> <given-names>OL</given-names>
</name>
</person-group>. <article-title>Breast cancer diagnosis <italic>via</italic> linear programming</article-title>. <source>IEEE Comput Sci Eng</source> (<year>1995</year>) <volume>2</volume>:<page-range>70&#x2013;1</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/MCSE.1995.414885</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jansen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Geleijnse</surname> <given-names>G</given-names>
</name>
<name>
<surname>van Maaren</surname> <given-names>M</given-names>
</name>
<name>
<surname>Hendriks</surname> <given-names>MP</given-names>
</name>
<name>
<surname>ten Teije</surname> <given-names>A</given-names>
</name>
<name>
<surname>Moncada-Torres</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Machine learning explainability in breast cancer survival. Pape-Haugaard LB, Lovis C, Madsen IC, et al (eds) Digital Personalized Health and Medicine. IOS Press, pp 307&#x2013;311</article-title> (<year>2020</year>), <page-range>307&#x2013;11</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.3233/SHTI200172</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pfob</surname> <given-names>A</given-names>
</name>
<name>
<surname>Mehrara</surname> <given-names>BJ</given-names>
</name>
<name>
<surname>Nelson</surname> <given-names>JA</given-names>
</name>
<name>
<surname>Wilkins</surname> <given-names>EG</given-names>
</name>
<name>
<surname>Pusic</surname> <given-names>AL</given-names>
</name>
<name>
<surname>Sidey-Gibbons</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Towards patient-centered decision-making in breast cancer surgery: Machine learning to predict individual patient-reported outcomes at 1-year follow-up</article-title>. <source>Ann Surg</source> (<year>2021</year>) <volume>277</volume>:<page-range>e144&#x2013;52</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1097/SLA.0000000000004862</pub-id>
</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>R</given-names>
</name>
<name>
<surname>Shinde</surname> <given-names>A</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>A</given-names>
</name>
<name>
<surname>Glaser</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lyou</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Yuh</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>Machine learning&#x2013;based interpretation and visualization of nonlinear interactions in prostate cancer survival</article-title>. <source>JCO Clin Cancer Inform</source> (<year>2020</year>) <volume>4</volume>, <page-range>637&#x2013;46</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/cci.20.00002</pub-id>
</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hassan</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>S-C</given-names>
</name>
<name>
<surname>Asaad</surname> <given-names>M</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Offodile</surname> <given-names>ACII</given-names>
</name>
<name>
<surname>Sidey-Gibbons</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Novel machine learning approach for the prediction of hernia recurrence, surgical complication, and 30-day readmission after abdominal wall reconstruction</article-title>. <source>J Am Coll Surg</source> (<year>2022</year>) <volume>234</volume>:<page-range>918&#x2013;27</page-range>. doi: <pub-id pub-id-type="doi">10.1097/XCS.0000000000000141</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="web">
<person-group person-group-type="author">
<collab>R Core Team</collab>
</person-group>. <source>R: A language and environment for statistical computing</source> (<year>2022</year>). Available at: <uri xlink:href="https://www.r-project.org/">https://www.r-project.org/</uri>.</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Kuhn</surname> <given-names>M</given-names>
</name>
</person-group>. <source>Caret: Classification and regression training</source> (<year>2022</year>). Available at: <uri xlink:href="https://cran.r-project.org/package=caret">https://cran.r-project.org/package=caret</uri>.</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Biecek</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>DALEX: Explainers for complex predictive models in R</article-title>. <source>J Mach Learn Res</source> (<year>2018</year>) <volume>19</volume>:<fpage>1</fpage>&#x2013;<lpage>5</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1806.08915</pub-id>
</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Hvitfeldt</surname> <given-names>E</given-names>
</name>
<name>
<surname>Pedersen</surname> <given-names>TL</given-names>
</name>
<name>
<surname>Benesty</surname> <given-names>M</given-names>
</name>
</person-group>. <source>Lime: Local interpretable model-agnostic explanations</source> (<year>2022</year>). Available at: <uri xlink:href="https://cran.r-project.org/package=lime">https://cran.r-project.org/package=lime</uri>.</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Kobyli&#x144;ska</surname> <given-names>K</given-names>
</name>
<name>
<surname>Miko&#x142;ajczyk</surname> <given-names>T</given-names>
</name>
<name>
<surname>Adamek</surname> <given-names>M</given-names>
</name>
<name>
<surname>Or&#x142;owski</surname> <given-names>T</given-names>
</name>
<name>
<surname>Biecek</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Explainable machine learning for modeling of early postoperative mortality in lung cancer</article-title>. In: <source>7th joint workshop on knowledge representation for health care and process-oriented information systems in health care, KR4HC/ProHealth 2019 and the 1st workshop on transparent, explainable and affective AI in medical systems, TEAAM 2019 held in conjuncti</source> (<publisher-loc>Poznan, Poland</publisher-loc>: <publisher-name>Springer, Cham</publisher-name>), vol. <volume>11979</volume>. (<year>2019</year>). p. <page-range>161&#x2013;74</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-37446-4_13</pub-id>
</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bertsimas</surname> <given-names>D</given-names>
</name>
<name>
<surname>Dunn</surname> <given-names>J</given-names>
</name>
<name>
<surname>Pawlowski</surname> <given-names>C</given-names>
</name>
<name>
<surname>Silberholz</surname> <given-names>J</given-names>
</name>
<name>
<surname>Weinstein</surname> <given-names>A</given-names>
</name>
<name>
<surname>Zhuo</surname> <given-names>YD</given-names>
</name>
<etal/>
</person-group>. <article-title>Applied informatics decision support tool for mortality predictions in patients with cancer</article-title>. <source>JCO Clin Cancer Inform</source> (<year>2018</year>) <volume>2</volume>:<fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/CCI.18.00003</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="web">
<person-group person-group-type="author">
<collab>U.S. Department of Health and Human Services Food and Drug Administration</collab>
</person-group>. <source>Good machine learning practice for medical device development: Guiding principles</source> (<year>2021</year>). Available at: <uri xlink:href="https://www.fda.gov/medical-devices/software-medical-device-samd/good-machine-learning-practice-medical-device-development-guiding-principles">https://www.fda.gov/medical-devices/software-medical-device-samd/good-machine-learning-practice-medical-device-development-guiding-principles</uri> (Accessed <access-date>December 8, 2022</access-date>).</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Demir</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>A decision support tool for predicting patients at risk of readmission: A comparison of classification trees, logistic regression, generalized additive models, and multivariate adaptive regression splines</article-title>. <source>Decision Sci</source> (<year>2014</year>) <volume>45</volume>:<page-range>849&#x2013;80</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/DECI.12094</pub-id>
</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname> <given-names>JH</given-names>
</name>
<name>
<surname>Roosen</surname> <given-names>CB</given-names>
</name>
</person-group>. <article-title>An introduction to multivariate adaptive regression splines</article-title>. <source>Stat Methods Med Res</source> (<year>1995</year>) <volume>4</volume>:<fpage>197</fpage>&#x2013;<lpage>217</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1177/096228029500400303</pub-id>
</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Boye</surname> <given-names>KS</given-names>
</name>
<name>
<surname>Lage</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Shinde</surname> <given-names>S</given-names>
</name>
<name>
<surname>Thieu</surname> <given-names>V</given-names>
</name>
<name>
<surname>Bae</surname> <given-names>JP</given-names>
</name>
</person-group>. <article-title>Trends in HbA1c and body mass index among individuals with type 2 diabetes: Evidence from a US database 2012&#x2013;2019</article-title>. <source>Diabetes Ther</source> (<year>2021</year>) <volume>12</volume>:<fpage>2077</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/S13300-021-01084-0</pub-id>
</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jiang</surname> <given-names>T</given-names>
</name>
<name>
<surname>Gradus</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Rosellini</surname> <given-names>AJ</given-names>
</name>
</person-group>. <article-title>Supervised machine learning: A brief primer</article-title>. <source>Behav Ther</source> (<year>2020</year>) <volume>51</volume>:<page-range>675&#x2013;87</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.BETH.2020.05.002</pub-id>
</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>SS</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>SS</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Pang</surname> <given-names>L</given-names>
</name>
<name>
<surname>Lam</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>Using the K-nearest neighbor algorithm for the classification of lymph node metastasis in gastric cancer</article-title>. <source>Comput Math Methods Med</source> (<year>2012</year>) <volume>2012</volume>:<elocation-id>11</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2012/876545</pub-id>
</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Palczewska</surname> <given-names>A</given-names>
</name>
<name>
<surname>Palczewski</surname> <given-names>J</given-names>
</name>
<name>
<surname>Marchese Robinson</surname> <given-names>R</given-names>
</name>
<name>
<surname>Neagu</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>Interpreting random forest classification models using a feature contribution method</article-title>. In: Bouabana-Tebibel, T., Rubin, S. (eds) <source>Integration of Reusable Systems. Advances in Intelligent Systems and Computing</source> (<year>2014</year>) (<publisher-loc>Springer</publisher-loc>: <publisher-name>Cham</publisher-name>) <volume>263</volume>, <fpage>193</fpage>&#x2013;<lpage>218</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-319-04717-1_9</pub-id>
</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Polato</surname> <given-names>M</given-names>
</name>
<name>
<surname>Aiolli</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>Boolean kernels for rule based interpretation of support vector machines</article-title>. <source>Neurocomputing</source> (<year>2019</year>) <volume>342</volume>:<page-range>113&#x2013;24</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.NEUCOM.2018.11.094</pub-id>
</citation>
</ref>
<ref id="B49">
<label>49</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Montavon</surname> <given-names>G</given-names>
</name>
<name>
<surname>Samek</surname> <given-names>W</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname> <given-names>KR</given-names>
</name>
</person-group>. <article-title>Methods for interpreting and understanding deep neural networks</article-title>. <source>Digital Signal Processing: A Rev J</source> (<year>2018</year>) <volume>73</volume>:<fpage>1</fpage>&#x2013;<lpage>15</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/J.DSP.2017.10.011</pub-id>
</citation>
</ref>
<ref id="B50">
<label>50</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hayashi</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Setiono</surname> <given-names>R</given-names>
</name>
<name>
<surname>Yoshida</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>A comparison between two neural network rule extraction techniques for the diagnosis of hepatobiliary disorders</article-title>. <source>Artif Intell Med</source> (<year>2000</year>) <volume>20</volume>:<page-range>205&#x2013;16</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/S0933-3657(00)00064-6</pub-id>
</citation>
</ref>
<ref id="B51">
<label>51</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Molnar</surname> <given-names>C</given-names>
</name>
<name>
<surname>Casalicchio</surname> <given-names>G</given-names>
</name>
<name>
<surname>Bischl</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Interpretable machine learning &#x2013; a brief history, state-of-the-art and challenges</article-title>. <source>Commun Comput Inf Sci</source> (<year>2020</year>) <volume>1323</volume>:<page-range>417&#x2013;31</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-65965-3_28</pub-id>
</citation>
</ref>
<ref id="B52">
<label>52</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Casalicchio</surname> <given-names>G</given-names>
</name>
<name>
<surname>Molnar</surname> <given-names>C</given-names>
</name>
<name>
<surname>Bischl</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Visualizing the feature importance for black box models</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Berlingerio</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bonchi</surname> <given-names>F</given-names>
</name>
<name>
<surname>G&#xe4;rtner</surname> <given-names>T</given-names>
</name>
<name>
<surname>Hurley</surname> <given-names>N</given-names>
</name>
<name>
<surname>Ifrim</surname> <given-names>G</given-names>
</name>
</person-group>, editors. <source>Machine learning and knowledge discovery in databases-European conference, ECML PKDD 2018</source>. <publisher-loc>Dublin, Ireland</publisher-loc>: <publisher-name>Springer Verlag</publisher-name> (<year>2019</year>). p. <page-range>655&#x2013;70</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-030-10925-7_40</pub-id>
</citation>
</ref>
<ref id="B53">
<label>53</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fisher</surname> <given-names>A</given-names>
</name>
<name>
<surname>Rudin</surname> <given-names>C</given-names>
</name>
<name>
<surname>Dominici</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>All models are wrong, but many are useful: Learning a variable&#x2019;s importance by studying an entire class of prediction models simultaneously</article-title>. <source>J Mach Learn Res</source> (<year>2019</year>) <volume>20</volume>:<fpage>1</fpage>&#x2013;<lpage>81</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1801.01489</pub-id>
</citation>
</ref>
<ref id="B54">
<label>54</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Iivanainen</surname> <given-names>S</given-names>
</name>
<name>
<surname>Ekstrom</surname> <given-names>J</given-names>
</name>
<name>
<surname>Virtanen</surname> <given-names>H</given-names>
</name>
<name>
<surname>v.</surname> <given-names>KV</given-names>
</name>
<name>
<surname>Koivunen</surname> <given-names>JP</given-names>
</name>
</person-group>. <article-title>Electronic patient-reported outcomes and machine learning in predicting immune-related adverse events of immune checkpoint inhibitor therapies</article-title>. <source>BMC Med Inform Decis Mak</source> (<year>2021</year>) <volume>21</volume>:<fpage>205</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12911-021-01564-0</pub-id>
</citation>
</ref>
<ref id="B55">
<label>55</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname> <given-names>JH</given-names>
</name>
</person-group>. <article-title>Greedy function approximation: A gradient boosting machine</article-title>(<year>2001</year>) (Accessed <access-date>October 18, 2022</access-date>).</citation>
</ref>
<ref id="B56">
<label>56</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Hastie</surname> <given-names>T</given-names>
</name>
</person-group>. <article-title>Causal interpretations of black-box models</article-title>. <source>J Business Economic Stat</source> (<year>2021</year>) <volume>39</volume>:<page-range>272&#x2013;81</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/07350015.2019.1624293</pub-id>
</citation>
</ref>
<ref id="B57">
<label>57</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Apley</surname> <given-names>DW</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Visualizing the effects of predictor variables in black box supervised learning models</article-title>. <source>J R Stat Soc Ser B Stat Methodol</source> (<year>2020</year>) <volume>82</volume>:<page-range>1059&#x2013;86</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/RSSB.12377</pub-id>
</citation>
</ref>
<ref id="B58">
<label>58</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Biecek</surname> <given-names>P</given-names>
</name>
<name>
<surname>Burzykowski</surname> <given-names>T</given-names>
</name>
</person-group>. <source>Explanatory model analysis</source> (<year>2021</year>). <publisher-loc>New York</publisher-loc>: <publisher-name>Chapman and Hall/CRC</publisher-name>. Available at: <uri xlink:href="https://ema.drwhy.ai/">https://ema.drwhy.ai/</uri> (Accessed <access-date>October 17, 2022</access-date>).</citation>
</ref>
<ref id="B59">
<label>59</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Staniak</surname> <given-names>M</given-names>
</name>
<name>
<surname>Biecek</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Explanations of model predictions with live and breakDown packages</article-title>. <source>R J</source> (<year>2019</year>) <volume>10</volume>:<fpage>395</fpage>&#x2013;<lpage>409</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.32614/RJ-2018-072</pub-id>
</citation>
</ref>
<ref id="B60">
<label>60</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>R</given-names>
</name>
<name>
<surname>Shinde</surname> <given-names>A</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>A</given-names>
</name>
<name>
<surname>Glaser</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lyou</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Yuh</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>Machine learning-based interpretation and visualization of nonlinear interactions in prostate cancer survival</article-title>. <source>JCO Clin Cancer Inform</source> (<year>2020</year>) <volume>4</volume>:<page-range>637&#x2013;46</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1200/cci.20.00002</pub-id>
</citation>
</ref>
<ref id="B61">
<label>61</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Goldstein</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kapelner</surname> <given-names>A</given-names>
</name>
<name>
<surname>Bleich</surname> <given-names>J</given-names>
</name>
<name>
<surname>Pitkin</surname> <given-names>E</given-names>
</name>
</person-group>. <article-title>Peeking inside the black box: visualizing statistical learning with plots of individual conditional expectation</article-title>. <source>J Comput Graphical Stat</source> (<year>2015</year>) <volume>24</volume>:<fpage>44</fpage>&#x2013;<lpage>65</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/10618600.2014.907095</pub-id>
</citation>
</ref>
<ref id="B62">
<label>62</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname> <given-names>K</given-names>
</name>
<name>
<surname>Ayyasamy</surname> <given-names>MV.</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Balachandran</surname> <given-names>PV.</given-names>
</name>
</person-group> <article-title>A comparison of explainable artificial intelligence methods in the phase classification of multi-principal element alloys</article-title>. <source>Sci Rep</source> (<year>2022</year>) <volume>12</volume>:<fpage>1</fpage>&#x2013;<lpage>15</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-022-15618-4</pub-id>
</citation>
</ref>
<ref id="B63">
<label>63</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rudin</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>Stop explaining black box machine learning models for high stakes decisions and use interpretable models instead</article-title>. <source>Nat Mach Intell</source> (<year>2019</year>) <volume>1</volume>:<page-range>206&#x2013;15</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42256-019-0048-x</pub-id>
</citation>
</ref>
<ref id="B64">
<label>64</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shaw</surname> <given-names>J</given-names>
</name>
<name>
<surname>Rudzicz</surname> <given-names>F</given-names>
</name>
<name>
<surname>Jamieson</surname> <given-names>T</given-names>
</name>
<name>
<surname>Goldfarb</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Artificial intelligence and the implementation challenge</article-title>. <source>J Med Internet Res</source> (<year>2019</year>) <volume>21</volume>:<fpage>e13659</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2196/13659</pub-id>
</citation>
</ref>
<ref id="B65">
<label>65</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Montani</surname> <given-names>S</given-names>
</name>
<name>
<surname>Striani</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Artificial intelligence in clinical decision support: A focused literature survey</article-title>. <source>Yearb Med Inform</source> (<year>2019</year>) <volume>28</volume>:<page-range>120&#x2013;7</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1055/s-0039-1677911</pub-id>
</citation>
</ref>
<ref id="B66">
<label>66</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Schwartz</surname> <given-names>R</given-names>
</name>
<name>
<surname>Flickinger</surname> <given-names>J</given-names>
</name>
<name>
<surname>Beriwal</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Machine learning approaches for predicting radiation therapy outcomes: A clinician&#x2019;s perspective</article-title>. <source>Int J Radiat Oncol Biol Phys</source> (<year>2015</year>) <volume>93</volume>:<page-range>1127&#x2013;35</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ijrobp.2015.07.2286</pub-id>
</citation>
</ref>
<ref id="B67">
<label>67</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kelly</surname> <given-names>CJ</given-names>
</name>
<name>
<surname>Karthikesalingam</surname> <given-names>A</given-names>
</name>
<name>
<surname>Suleyman</surname> <given-names>M</given-names>
</name>
<name>
<surname>Corrado</surname> <given-names>G</given-names>
</name>
<name>
<surname>King</surname> <given-names>D</given-names>
</name>
</person-group>. <article-title>Key challenges for delivering clinical impact with artificial intelligence</article-title>. <source>BMC Med</source> (<year>2019</year>) <volume>17</volume>:<fpage>195</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12916-019-1426-2</pub-id>
</citation>
</ref>
<ref id="B68">
<label>68</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Molnar</surname> <given-names>C</given-names>
</name>
<name>
<surname>K&#xf6;nig</surname> <given-names>G</given-names>
</name>
<name>
<surname>Herbinger</surname> <given-names>J</given-names>
</name>
<name>
<surname>Freiesleben</surname> <given-names>T</given-names>
</name>
<name>
<surname>Dandl</surname> <given-names>S</given-names>
</name>
<name>
<surname>Scholbeck</surname> <given-names>CA</given-names>
</name>
<etal/>
</person-group>. <article-title>General pitfalls of model-agnostic interpretation methods for machine learning models</article-title>. In: <person-group person-group-type="editor">
<name>
<surname>Holzinger</surname> <given-names>A</given-names>
</name>
<name>
<surname>Goebel</surname> <given-names>R</given-names>
</name>
<name>
<surname>Fong</surname> <given-names>R</given-names>
</name>
<name>
<surname>Moon</surname> <given-names>T</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname> <given-names>K-R</given-names>
</name>
<name>
<surname>Samek</surname> <given-names>W</given-names>
</name>
</person-group>, editors. <source>xxAI - beyond explainable AI: International workshop, held in conjunction with ICML 2020, July 18, 2020, Vienna, Austria, revised and extended papers</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer International Publishing</publisher-name> (<year>2022</year>). p. <fpage>39</fpage>&#x2013;<lpage>68</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-031-04083-2_4</pub-id>
</citation>
</ref>
<ref id="B69">
<label>69</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>A</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>F</given-names>
</name>
<name>
<surname>Gu</surname> <given-names>F</given-names>
</name>
<name>
<surname>Gama</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>G</given-names>
</name>
</person-group>. <article-title>Learning under concept drift: A review</article-title>. <source>IEEE Trans Knowl Data Eng</source> (<year>2019</year>) <volume>31</volume>:<page-range>2346&#x2013;63</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TKDE.2018.2876857</pub-id>
</citation>
</ref>
<ref id="B70">
<label>70</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Finlayson</surname> <given-names>SG</given-names>
</name>
<name>
<surname>Subbaswamy</surname> <given-names>A</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>K</given-names>
</name>
<name>
<surname>Bowers</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kupke</surname> <given-names>A</given-names>
</name>
<name>
<surname>Zittrain</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>The clinician and dataset shift in artificial intelligence</article-title>. <source>New Engl J Med</source> (<year>2021</year>) <volume>385</volume>:<page-range>283&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1056/nejmc2104626</pub-id>
</citation>
</ref>
<ref id="B71">
<label>71</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duckworth</surname> <given-names>C</given-names>
</name>
<name>
<surname>Chmiel</surname> <given-names>FP</given-names>
</name>
<name>
<surname>Burns</surname> <given-names>DK</given-names>
</name>
<name>
<surname>Zlatev</surname> <given-names>ZD</given-names>
</name>
<name>
<surname>White</surname> <given-names>NM</given-names>
</name>
<name>
<surname>Daniels</surname> <given-names>TWV</given-names>
</name>
<etal/>
</person-group>. <article-title>Using explainable machine learning to characterise data drift and detect emergent health risks for emergency department admissions during COVID-19</article-title>. <source>Sci Rep</source> (<year>2021</year>) <volume>11</volume>:<fpage>1</fpage>&#x2013;<lpage>10</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-021-02481-y</pub-id>
</citation>
</ref>
<ref id="B72">
<label>72</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Slack</surname> <given-names>D</given-names>
</name>
<name>
<surname>Hilgard</surname> <given-names>S</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>E</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lakkaraju</surname> <given-names>H</given-names>
</name>
</person-group>. (<year>2020</year>). <article-title>Fooling LIME and SHAP: Adversarial attacks on post hoc explanation methods</article-title>, in: <conf-name>AIES &#x2018;20: Proceedings of the AAAI/ACM Conference on AI, Ethics, and Society</conf-name>, <conf-loc>New York, NY</conf-loc> (<publisher-loc>New York, NY</publisher-loc>: <publisher-name>Association for Computing Machinery (ACM)</publisher-name>). pp. <page-range>180&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1145/3375627.3375830</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>