<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Astron. Space Sci.</journal-id>
<journal-title>Frontiers in Astronomy and Space Sciences</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Astron. Space Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-987X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">1651953</article-id>
<article-id pub-id-type="doi">10.3389/fspas.2025.1651953</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Astronomy and Space Sciences</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Local-NPDR: a novel variable importance method for explainable machine learning and false discovery diagnosis for ocean worlds biosignatures</article-title>
<alt-title alt-title-type="left-running-head">Clough et al.</alt-title>
<alt-title alt-title-type="right-running-head">
<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.3389/fspas.2025.1651953">10.3389/fspas.2025.1651953</ext-link>
</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Clough</surname>
<given-names>Lily A.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3184780/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Major</surname>
<given-names>Jonathan D.</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2243913/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Seyler</surname>
<given-names>Lauren M.</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/618021/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Da Poian</surname>
<given-names>Victoria</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1447306/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Theiling</surname>
<given-names>Bethany P.</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1496869/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>McKinney</surname>
<given-names>Brett A.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/33363/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/Writing - review &#x26; editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Tandy School of Computer Science, The University of Tulsa</institution>, <addr-line>Tulsa</addr-line>, <addr-line>OK</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Aurora Engineering</institution>, <addr-line>Reston</addr-line>, <addr-line>VA</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Planetary Environments Laboratory, NASA Goddard Space Flight Center</institution>, <addr-line>Greenbelt</addr-line>, <addr-line>MD</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>School of Geosciences, University of South Florida</institution>, <addr-line>Tampa</addr-line>, <addr-line>FL</addr-line>, <country>United States</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>School of Natural Sciences and Mathematics, Stockton University</institution>, <addr-line>Galloway</addr-line>, <addr-line>NJ</addr-line>, <country>United States</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Earth and Planetary Science, Johns Hopkins University</institution>, <addr-line>Baltimore</addr-line>, <addr-line>MD</addr-line>, <country>United States</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Tyto Athene LLC</institution>, <addr-line>Reston</addr-line>, <addr-line>VA</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/103559/overview">Josep M. Trigo-Rodr&#xed;guez</ext-link>, Spanish National Research Council (CSIC), Spain</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2385849/overview">Pierfrancesco Novielli</ext-link>, University of Bari Aldo Moro, Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3116791/overview">Floyd Nichols</ext-link>, Virginia Tech, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Brett A. McKinney, <email>brett.mckinney@gmail.com</email>
</corresp>
</author-notes>
<pub-date pub-type="epub">
<day>24</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1651953</elocation-id>
<history>
<date date-type="received">
<day>22</day>
<month>06</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Clough, Major, Seyler, Da Poian, Theiling and McKinney.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Clough, Major, Seyler, Da Poian, Theiling and McKinney</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Explainable machine learning (ML) is important for biosignature prediction on future astrobiology missions to minimize the risk of false positives due to geochemical biotic mimicry and false negatives due to environmental factors that mask biosignatures. ML models often use feature importance scores to provide insights into model prediction mechanisms by quantifying each variable&#x2019;s contribution to the prediction. Global variable importance methods aggregate information across all training samples and therefore do not provide interpretation for the classification of a single sample. In contrast, local variable importance scores quantify the contribution of variables to the classification of a single sample and can therefore help explain why the sample was predicted to be in a certain class and diagnose whether it is a false prediction. We present a new local variable importance method that handles nonlinearity, statistical interactions, and includes penalized feature selection. Our approach represents a local version of Nearest-neighbor Projected Distance Regression (NPDR) feature selection. We evaluate local-NPDR on complex simulated data and real data from a study of carbon and oxygen isotopic biosignatures using laboratory-generated ocean world analogue brines. The ability of local-NPDR to differentiate between true and false predictions is compared with other common local importance methods. Local-NPDR is able to diagnose individual false predictions using the concordance between global and local scores, and it can explain mechanisms of true and false predictions. These features allow local-NPDR to integrate scientific explanations of single-sample ML predictions to support a more comprehensive framework for biosignature detection.</p>
</abstract>
<kwd-group>
<kwd>explainable machine learning</kwd>
<kwd>biosignature detection</kwd>
<kwd>local importance scores</kwd>
<kwd>ocean worlds</kwd>
<kwd>biotic mimicry</kwd>
</kwd-group>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Astrobiology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<disp-quote>
<p>&#x201c;It is the desire for explanations which are at once systematic and controllable by factual evidence that generates science; and it is the organization and classification of knowledge on the basis of explanatory principles that is the distinctive goal of the sciences.&#x201d;</p>
</disp-quote>
<disp-quote>
<p>&#x2013;Ernest Nagel, The Structure of Science</p>
</disp-quote>
<p>Machine learning (ML) has become a widespread tool for scientific data analysis and is increasingly used in hybrid modeling to predict physical processes (<xref ref-type="bibr" rid="B20">Noordijk et al., 2024</xref>). In a utilitarian sense, the goals of ML and science parallel each other: both seek to make accurate and practical predictions. Science and ML achieve this by finding generalizable regularities in data that can be used for prediction. In science, these regularities may become elevated to the status of a law, which is a distillation of complex data in a form that services another, deeper goal: explanation or understanding. Indeed, one of the most important goals of science is to make nature intelligible to humans by providing insights into the mechanisms by which natural phenomena occur; <italic>i.e.</italic>, to provide scientific explanations (<xref ref-type="bibr" rid="B17">Nagel, 1979</xref>). For ML, regularities found in data are encapsulated in a statistical model or algorithm; however, ML model predictions usually cannot be easily explained like a scientific law and are often likened to &#x201c;black boxes&#x201d;.</p>
<p>Increasingly in critical and high-risk domains, the need for transparency, explainability, and interpretability in ML model predictions has been recognized (<xref ref-type="bibr" rid="B11">Linardatos et al., 2020</xref>; <xref ref-type="bibr" rid="B24">Roscher et al., 2020</xref>). Transparency means that the algorithmic mechanisms and parameter space through which ML predictions are made are understandable and reproducible, while interpretability refers to the ability to draw connections between model predictions and the scientific domain to be understood (<xref ref-type="bibr" rid="B16">Montavon et al., 2018</xref>; <xref ref-type="bibr" rid="B24">Roscher et al., 2020</xref>). Explainability can be defined as a highly relevant feature (variable) space (<xref ref-type="bibr" rid="B24">Roscher et al., 2020</xref>) through which interpretations about model predictions can be made. To add explainability to ML model predictions, a variable space that is mathematically (and if possible, physically) understandable can be leveraged to connect the variables (and their physical/mathematical meanings) to the particular ML predictions being made and to the scientific problem at hand (<italic>i.e.</italic>, interpretation). These tools provide a series of explanatory principles upon which the ML model prediction can be understood. In this way, trust and explainability in ML are inextricable, as they are in science.</p>
<p>Astrobiology offers an enticing problem for ML: how can we accurately detect the presence of life in an unknown environment of unknown history? The ability to trust an ML model prediction is crucial for such a high-risk scientific question. In the remote locations of proposed astrobiological targets, it is not possible to directly verify whether a biosignature prediction is true or false. Therefore, explanatory false detection tools will be necessary for astrobiology missions. False positive (FP) and false negative (FN) biosignature detection using remote sensing is a well-documented challenge (<xref ref-type="bibr" rid="B18">National Academies of Sciences, Engineering, and Medicine, 2019</xref>). Abiotic environments with complex geochemistry can mimic a biosignature, leading to a FP, or the environment can mask a biosignature prediction, leading to a FN (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>). Autonomous decision making based on ML and artificial intelligence (AI) can make space missions more efficient, but the risk of false predictions must be mitigated, both to protect mission resources and to instill trust in real-time ML analysis of collected data (<xref ref-type="bibr" rid="B25">Theiling et al., 2022</xref>; <xref ref-type="bibr" rid="B5">Da Poian et al., 2025</xref>). These examples underscore the importance of interpreting ML predictions in the context of the geochemical environment, using training data that accurately reflects the target deployment environment, and diagnosing false predictions.</p>
<p>Although not universally the case, ML tends to suffer from an accuracy-explainability tradeoff (<xref ref-type="bibr" rid="B1">Ali et al., 2023</xref>). As data dimensionality (number of features) has increased across research fields, ML models have improved in accuracy but grown in complexity, often resulting in &#x201c;black box&#x201d; systems with limited transparency of their decision-making process and relevant predictors (variables). This high-dimensionality and increased opacity in algorithmic mechanisms results in decreased explainability. For scientific models, explanation is often built into the model in terms of the mathematical symbols that describe physical laws. In this way scientific and ML models have different levels of inherent transparency and explainability. The most transparent model is one whose exact mechanism for prediction is comprehensible to a human. For example, a decision tree model has a high level of transparency (<italic>i.e.,</italic> a &#x201c;transparent box&#x201d;). Its decision-making process can be followed for each variable split in the tree for a given sample, and the structure of the tree gives some explainability as well: nodes (variables) at the top have the highest variable importance and branches connecting variables may suggest conditional relationships. Unfortunately, its prediction accuracy is not high enough in most applications, which led to resampling methods like Random Forest (RF) (<xref ref-type="bibr" rid="B2">Breiman, 2001</xref>). The many trees (forest) used by RF to vote on sample classes is responsible for its improved accuracy but also reduces its explainability.</p>
<p>ML tools can provide global and/or local explainability; global explainability results from generalizations made across all training samples, while local explainability focuses on one sample or a neighborhood of samples (<xref ref-type="bibr" rid="B24">Roscher et al., 2020</xref>). While RF is on the opaque end of the transparency spectrum, it does provide tools for global and local explainability such as permutation variable importance (<xref ref-type="bibr" rid="B2">Breiman, 2001</xref>). For an important variable, the permutation importance score increases if the out-of-bag (oob) accuracy of the model decreases after permuting the variable. Permutation importance thus provides a degree of explanation by ranking which variables the RF model finds most necessary for prediction. This importance method is global in that it aggregates information across all training samples and the scores are not specific to explaining an individual sample&#x2019;s prediction. To address this, RF has a local version of permutation importance that gives variable importance scores specific to the prediction for each sample in the training data.</p>
<p>Recently, we showed that local (single-sample) RF variable importance has the potential to add to the explainability of ML biosignature model predictions and can help diagnose false predictions (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>). We used our Nearest-neighbors Projected Distance Regression (NPDR) global feature selection with an RF classifier for biosignature prediction (<xref ref-type="bibr" rid="B10">Le et al., 2020</xref>), and we computed the discordance between global and local scores for single samples to diagnose false predictions. This global-local discordance provided important insights; however, existing local variable (or feature) importance methods face limitations such as needing samples to be in the training data, the lack of a statistical threshold for feature selection, and limited ability to account for statistical interactions. A statistical interaction occurs when the effect of a feature on the outcome variable depends on one or more other features (<xref ref-type="bibr" rid="B13">McKinney et al., 2006</xref>). An example of an interaction is conditional correlation between pairs of variables that depends on the outcome variable and may occur without either variable having a main effect. These interactions are likely to be important for uncovering biotic mimicry (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>), and therefore techniques for evaluating the probability of true/false positive or negative predictions of biosignatures are needed for future astrobiology missions.</p>
<p>In high-dimensional variable spaces expected for data of astrobiological relevance (e.g., mass spectrometry and spectroscopy), RF has low power to detect statistical interactions between variables that may be important for classification because variables are selected in trees preferentially based on main effects (<xref ref-type="bibr" rid="B14">McKinney et al., 2009</xref>; <xref ref-type="bibr" rid="B27">Wright et al., 2016</xref>). However, our global feature selection method, NPDR, has shown high power to detect both main effects and statistical interactions in the biosignature model, and it uses a regression penalty to yield a reduced dimensionality space of independent features (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>). In the current study, we present an additional mechanism for evaluating the reliability of biosignature predictions. We extend NPDR to compute local or single-sample importance scores to take advantage of NPDR&#x2019;s ability to detect complex relationships between variables. Our global variable importance ML tool helps satisfy the deeper goal of science to provide explanations by allowing an ML model to be trained in a selected features space that is as relevant as possible to the outcome being analyzed. Our local feature importance ML tool introduced in the current study provides explanatory analysis for a single sample in the context of globally important variables and a given ML prediction. Crucially, this analysis allows for a determination of whether a single sample prediction is likely to be true or false without knowing the actual sample label.</p>
<p>The critical need for explainable ML methods across multiple disciplines has given rise to additional local methods since the advent of local-RF, such as Local Interpretable Model-Agnostic Explanations (LIME) and SHapley Additive exPlanations (SHAP). LIME creates local importance scores for models by fitting a surrogate linear model to synthetic samples in the local neighborhood of a query sample (<xref ref-type="bibr" rid="B23">Ribeiro et al., 2016</xref>). SHAP uses game theory concepts to provide model agnostic local scores that contribute to the prediction of a sample, while TreeSHAP provides local feature explanations specifically for tree based models like RF (<xref ref-type="bibr" rid="B12">Lundberg et al., 2020</xref>). These core tools have been used in a variety of scientific and medical domains including those related to geosciences such as physical oceanography (<xref ref-type="bibr" rid="B19">Navarra et al., 2025</xref>), flood susceptibility (<xref ref-type="bibr" rid="B3">Choubin et al., 2025</xref>), and CO<sub>2</sub> changes in the soil (<xref ref-type="bibr" rid="B21">Novielli et al., 2025</xref>). We compare local-NPDR and local-RF with these prominent methods.</p>
<p>The remainder of the manuscript is organized as follows. We describe the new local-NPDR method for single-sample variable importance, and we describe the simulated and real biosignature data. We compare local-NPDR with other local feature importance methods for the simulated data and real biosignature laboratory data based on the ability to explain and detect false predictions. The local-NPDR method is not specific to a given (ML) classifier, and it is able to model nonlinear decision boundaries and detect statistical interactions between features. We use local scores to explain which features a classifier might find most important for classifying a specific sample, and we use the discordance between global and local scores combined with single-sample prediction probabilities to flag potential false predictions. We then discuss the necessity of explainability and false prediction assessment for high-stakes predictions such as astrobiology biosignatures.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<p>In this section, we first describe the local-NPDR algorithm and formalism in the context of global-NPDR feature selection along with an illustration of its use for diagnosing true and false predictions. Next, we describe the procedure for designating a prediction as likely true or false based on the total local scores for concordance (positive) and discordance (negative) for globally important features. Finally, we describe the simulated and real biosignature datasets for validation of the local-NPDR algorithm.</p>
<sec id="s2-1">
<title>2.1 Local-NPDR: feature importance for a single sample</title>
<p>Consider a pair of samples or neighbors <italic>i</italic> and <italic>j</italic> that are distinct rows of an <italic>m</italic> x <italic>p</italic> data matrix <bold>X</bold> with <italic>m</italic> samples and <italic>p</italic> variables. The class vector <italic>y</italic> has length <italic>m</italic>. To describe the NPDR contrastive loss, we use the contribution to the binary cross-entropy for a pair of neighbors given a set of regression coefficients represented by <inline-formula id="inf1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>,<disp-formula id="e1">
<mml:math id="m2">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(1)</label>
</disp-formula>where <inline-formula id="inf2">
<mml:math id="m3">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the hit/miss indicator variable and <inline-formula id="inf3">
<mml:math id="m4">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted probability that the two samples are in different classes (<italic>e.g.</italic>, for the probability of a miss, <inline-formula id="inf4">
<mml:math id="m5">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). The indicator variable can have two values: <inline-formula id="inf5">
<mml:math id="m6">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if the pair of samples are in a different class (<inline-formula id="inf6">
<mml:math id="m7">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2260;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) and <inline-formula id="inf7">
<mml:math id="m8">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> if they are in the same class (<inline-formula id="inf8">
<mml:math id="m9">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mo>&#x3d;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>j</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>). The predicted probability <xref ref-type="disp-formula" rid="e2">Equation 2</xref> is computed using the following logit transformation<disp-formula id="e2">
<mml:math id="m10">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2b;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mover accent="true">
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
<label>(2)</label>
</disp-formula>of the multivariate model of projected distances, <inline-formula id="inf9">
<mml:math id="m11">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, of all independent variables in X. In other words, for a fixed pair of <italic>ij</italic> neighbors, each element of the vector, <inline-formula id="inf10">
<mml:math id="m12">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, is an absolute difference between their values for each independent variable in X. We refer to these differences as projected distances onto a variable axis in the <italic>p</italic>-dimensional space. For example, if X were a numeric data matrix, the vector of projected distances (<xref ref-type="disp-formula" rid="e3">Equation 3</xref>) would be<disp-formula id="e3">
<mml:math id="m13">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>X</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#x22ef;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mrow>
<mml:mfenced open="|" close="|" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(3)</label>
</disp-formula>
</p>
<p>The goal of local-NPDR is to find the variable importance scores (<inline-formula id="inf11">
<mml:math id="m14">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula>) that minimize the penalized negative log-likelihood (or cross entropy) over the neighborhood <inline-formula id="inf12">
<mml:math id="m15">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> of sample <italic>i</italic>
<disp-formula id="e4">
<mml:math id="m16">
<mml:mrow>
<mml:msubsup>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mi mathvariant="italic">min</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
<label>(4)</label>
</disp-formula>
</p>
<p>The penalty is implemented via the R library <italic>glmnet</italic>, which allows for a blend of L<sub>1</sub> and L<sub>2</sub> regularization in a method called the &#x201c;elastic net&#x201d; (<xref ref-type="bibr" rid="B26">Tibshirani, 1996</xref>), and it can be used for Ridge (L<sub>2</sub>, <inline-formula id="inf13">
<mml:math id="m17">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>) or LASSO (L<sub>1</sub>, <inline-formula id="inf14">
<mml:math id="m18">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>) penalized regression. We typically use LASSO (Least Absolute Shrinkage and Selection Operator) for global-NPDR feature selection and tune <inline-formula id="inf15">
<mml:math id="m19">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> via cross-validation. This reduces the selected feature space and increases variable independence. The nature of the derivative of the absolute value function in LASSO prevents regression coefficients from shrinking further once reaching zero as <inline-formula id="inf16">
<mml:math id="m20">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> increases, but rather they stay zero. For local-NPDR, we typically employ a Ridge penalty because we have already reduced the feature space using global-NPDR and want to keep the rankings of all selected features. The quantity <inline-formula id="inf17">
<mml:math id="m21">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the set of neighbors of sample <italic>i</italic>, and the resulting NPDR attribute scores, <inline-formula id="inf18">
<mml:math id="m22">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi mathvariant="normal">&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> are local to each sample <italic>i</italic>. The neighborhood is computed independently of the class status of samples and is defined using a distance matrix, discussed more below. These local variable importance scores indicate the importance of features that allow the single sample to discriminate whether neighbor samples are in the same or a different class as the target sample. If a variable were involved in an interaction, NPDR would reflect this in the importance score because it uses nearest neighbors that are computed in the higher dimensional space of all other variables. This makes NPDR multivariate even when scoring a single variable for a single sample. The Ridge or LASSO version of NPDR includes additional multivariate effects in its model.</p>
<p>We illustrate how local-NPDR feature selection can add support for true ML predictions of single samples (blue box sample 1, <xref ref-type="fig" rid="F1">Figure 1a</xref>) and can help identify false positive predictions (red box sample 1, <xref ref-type="fig" rid="F2">Figure 2a</xref>) by comparing the local score for a globally important variable (purple variable <italic>A</italic> on the vertical axis, <xref ref-type="fig" rid="F1">Figure 1</xref>). The global importance can be determined using global-NPDR. For completeness, the global NPDR scores are computed by minimizing the following penalized cross entropy<disp-formula id="e5">
<mml:math id="m23">
<mml:mrow>
<mml:msup>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x3d;</mml:mo>
<mml:munder>
<mml:mi mathvariant="italic">min</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:munderover>
</mml:mstyle>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>j</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:munder>
</mml:mstyle>
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b2;</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2b;</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>&#x3b1;</mml:mi>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mfenced open="&#x2016;" close="&#x2016;" separators="|">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x3b2;</mml:mi>
<mml:mo>&#x21c0;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
<label>(5)</label>
</disp-formula>which, in contrast to <xref ref-type="disp-formula" rid="e4">Equation 4</xref>, includes the sum over all samples <italic>i</italic> from 1 to <italic>m</italic>.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Local-NPDR mathematics for a correct classification with a positive (supporting) local feature importance score. <bold>(a)</bold> Consider hypothetical Sample-1 of class <italic>&#x2018;x&#x2019;</italic> (blue highlighted <bold>x</bold>
<sub>1</sub>) and two features, one simulated with a main effect for classification, <italic>i.e.</italic>, for discriminating between class <italic>&#x2018;x&#x2019;</italic> and <italic>&#x2018;o&#x2019;</italic> samples (variable <italic>A</italic>, purple) and one unimportant variable with no effect. The nearest neighbors for Sample-1, indicated inside the dashed neighborhood circle, are Sample-2 (<bold>x</bold>
<sub>2</sub>, same class as Sample-1) and Sample-3 (<bold>o</bold>
<sub>3</sub>, different class than Sample-1). The projected distances between Samples-1 and 2 for variable <italic>A</italic> (d<sub>12</sub>(A)) and Samples-1 and 3 (d<sub>13</sub>(A)) are also indicated by &#x394;<sub>H</sub> (hit) and &#x394;<sub>M</sub> (miss) because their actual hit/miss statuses are &#x3b4;<sub>12</sub> &#x3d; 0 (hit) and &#x3b4;<sub>13</sub> &#x3d; 1 (miss). Note that the projected distances for these same samples onto the horizontal axis (unimportant variable) are negligible because this variable cannot discriminate between samples in class <italic>&#x2018;x&#x2019;</italic> or class <italic>&#x2018;o&#x2019;</italic>. <bold>(b)</bold> Local-NPDR loss function for the two nearest neighbors of Sample-1 (see <xref ref-type="disp-formula" rid="e1">Equation 1</xref>). Note the total loss for variable <italic>A</italic> (for Sample-1) is the sum of all pairwise loss functions for all local neighbors. The loss functions for two pairs of neighbors (<inline-formula id="inf19">
<mml:math id="m24">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mn>12</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula id="inf20">
<mml:math id="m25">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mn>13</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) are small because Sample-1 is correctly classified and the &#x3b4;&#x2019;s are correctly assigned as &#x3b4;<sub>12</sub> &#x3d; 0 (hit) and &#x3b4;<sub>13</sub> &#x3d; 1 (miss). <bold>(c)</bold> These low losses for the true classification of Sample-1 as class <italic>&#x2018;x&#x2019;</italic> lead to positive local scores for important variable <italic>A</italic> in agreement with the global score.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g001.tif">
<alt-text content-type="machine-generated">Diagram illustrating the positive/supporting local score for a true prediction \( x_1 \). Panel (a) shows variables and distances \( d_{12}(A) \) and \( d_{13}(A) \) in an XY plot, with \( x_1 \) as the main point. Panel (b) details a loss function equation and describes scenarios for small and large distances leading to small loss and positive scores. Panel (c) displays a chart with variable A's importance, comparing global and sample 1 scores, noting that a high local score supports the prediction.</alt-text>
</graphic>
</fig>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Local-NPDR mathematics for an incorrect classification with a negative (contradicting) local feature importance score. <bold>(a)</bold> Same as <xref ref-type="fig" rid="F1">Figure 1</xref> except hypothetical Sample-1 is now incorrectly assigned class <italic>&#x2018;o&#x2019;</italic> (red highlighted <bold>o</bold>
<sub>1</sub>). Two features are simulated, one with a main effect for classification (variable <italic>A</italic>, purple) and one unimportant variable with no effect. The two nearest neighbors for Sample-1, indicated by the neighborhood circle, are Sample-2 (<bold>x</bold>
<sub>2</sub>, different class as Sample-1) and Sample-3 (<bold>o</bold>
<sub>3</sub>, same class as Sample-1). The projected distances between Samples-1 and 2 for variable <italic>A</italic> (d<sub>12</sub>(A)) and Samples-1 and 3 (d<sub>13</sub>(A)) are indicated by &#x394;<sub>M</sub> (miss) and &#x394;<sub>H</sub> (hit)because &#x3b4;<sub>12</sub> &#x3d; 1 (miss) and &#x3b4;<sub>13</sub> &#x3d; 0 (hit). <bold>(b)</bold> Local-NPDR loss function for the two nearest neighbors of Sample 1 (see <xref ref-type="disp-formula" rid="e1">Equation 1</xref>). The loss functions for two pairs of neighbors (<inline-formula id="inf21">
<mml:math id="m26">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mn>12</mml:mn>
</mml:msub>
<mml:mtext>&#x2009;</mml:mtext>
<mml:mi>a</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>d</mml:mi>
<mml:mtext>&#x2009;</mml:mtext>
<mml:msub>
<mml:mi mathvariant="script">L</mml:mi>
<mml:mn>13</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) are large because Sample 1 is incorrectly classified and the &#x3b4;&#x2019;s are incorrectly assigned as &#x3b4;<sub>12</sub> &#x3d; 1 (miss) and &#x3b4;<sub>13</sub> &#x3d; 0 (hit). <bold>(c)</bold> These large losses for the false classification of Sample-1 as class &#x2018;<italic>o&#x2019;</italic> lead to negative local importance scores for variable <italic>A</italic>.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g002.tif">
<alt-text content-type="machine-generated">Diagram illustrating a negative/contradicting local score for a false prediction, showing a misclassification of Sample 1 as o1. Panel a presents a plot with variables, distances \(d_{12}(A)\) and \(d_{13}(A)\), and indicators &#x3B4;. Panel b details a loss function or cross-entropy explanation with equations, indicating large losses and negative scores for specific probabilities. Panel c shows a comparison of the global score and Sample 1&#x2019;s contradicting impact, highlighting &#x394; conditions leading to a negative local score.</alt-text>
</graphic>
</fig>
<p>The fact that the purple variable <italic>A</italic> is globally important for classification can be seen by noticing that its mean for the &#x2018;x&#x2019; class is larger than its mean for the &#x2018;o&#x2019; class (<xref ref-type="fig" rid="F1">Figure 1</xref>). Note that in contrastive feature selection methods such as NPDR, a sample contributes positively to a variable&#x2019;s importance score if the projected distance along that variable axis to its opposite-class nearest neighbor (<inline-formula id="inf22">
<mml:math id="m27">
<mml:mrow>
<mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:mi>M</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, delta miss) is greater than the projected distance to its same-class nearest neighbor (<inline-formula id="inf23">
<mml:math id="m28">
<mml:mrow>
<mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, delta hit). This differential <inline-formula id="inf24">
<mml:math id="m29">
<mml:mrow>
<mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:mi>M</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mo>&#x394;</mml:mo>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> quantifies how well the variable keeps hits close together and misses farther apart in a neighborhood (<xref ref-type="bibr" rid="B15">McKinney et al., 2013</xref>).</p>
<p>First we consider how the globally important variable is affected locally in the local-NPDR contrastive loss (<xref ref-type="disp-formula" rid="e1">Equation 1</xref>) for Sample-1 when the sample is in the correct &#x2018;x&#x2019; class (x<sub>1</sub> in blue box, <xref ref-type="fig" rid="F1">Figure 1a</xref>). Specifically, we estimate the contributions to the contrastive loss (<xref ref-type="disp-formula" rid="e1">Equation 1</xref>) for variable <italic>A</italic> and Sample-1 using <italic>k</italic> &#x3d; 2 neighbors. In this case, the two neighbors are Sample-2 (same class as Sample 1 (hit): &#x2018;x&#x2019;) and Sample-3 (opposite class of Sample 1 (miss): &#x2018;o&#x2019;). The neighbor-pair loss for the miss L<sub>12</sub> is low (good fit) because the projected distance d<sub>12</sub> is small (<xref ref-type="fig" rid="F1">Figure 1b</xref>) leading to a small <inline-formula id="inf25">
<mml:math id="m30">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mn>12</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> miss-probability, and their actual miss state is <inline-formula id="inf26">
<mml:math id="m31">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mn>12</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, which causes the first term to be zero. That is, the non-zero quantity <inline-formula id="inf27">
<mml:math id="m32">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mn>12</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> will be a small positive loss (good fit), and the contribution to the local score from A for Sample-1would be relatively large. The neighbor-pair loss for the hit L<sub>13</sub> is also low (good fit) because, while d<sub>13</sub> is large, their actual miss state is <inline-formula id="inf28">
<mml:math id="m33">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mn>13</mml:mn>
</mml:msub>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, causing the second term to be zero. The remaining non-zero part of the loss <inline-formula id="inf29">
<mml:math id="m34">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b4;</mml:mi>
<mml:mn>13</mml:mn>
</mml:msub>
<mml:mo>&#x2061;</mml:mo>
<mml:mi>ln</mml:mi>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>d</mml:mi>
<mml:mo>&#x5e;</mml:mo>
</mml:mover>
<mml:mn>13</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>A</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> will be a small positive quantity, and the contribution of Sample-1 and Sample-3 neighbors to the local score for variable <italic>A</italic> will be relatively large. This high importance score in the local neighborhood of sample x<sub>1</sub> for the globally important variable A (concordant local and global scores) is supporting evidence that sample x<sub>1</sub> is a true positive (<xref ref-type="fig" rid="F1">Figure 1c</xref>). In contrast, if we incorrectly label Sample-1 (&#x2018;o&#x2019; instead of &#x2018;x&#x2019; in <xref ref-type="fig" rid="F2">Figure 2A</xref>), the neighbor losses will be high (bad fit, <xref ref-type="fig" rid="F2">Figure 2b</xref>) and the importance of variable <italic>A</italic> in the local neighborhood of Sample-1 will be low (<xref ref-type="fig" rid="F2">Figure 2c</xref>), discordant with the global importance of <italic>A</italic> and suggesting that Sample-1 might be a false prediction.</p>
<p>The quantity <italic>k</italic> in <inline-formula id="inf30">
<mml:math id="m35">
<mml:mrow>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mfenced open="(" close=")" separators="|">
<mml:mrow>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is the number of nearest neighbors used for sample <italic>i</italic> in NPDR, sometimes referred to as <italic>knn</italic> (<italic>k-</italic>nearest neighbors). This number can vary from sample to sample or be uniform (same for all samples). For global-NPDR <italic>knn</italic>, we use k&#x2013;D&#x3c3;<sub>1/2</sub>, which is the expected number of neighbors that are within &#xbd; standard deviation of the mean distance (D&#x3c3;<sub>1/2</sub>) between all sample pairs. For local-NPDR, which focuses on only one sample, we use <inline-formula id="inf31">
<mml:math id="m36">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>_</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="italic">max</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> because it maximizes the statistical power by using all possible samples in the neighborhood of the single sample. The tradeoff is a decreased ability to detect statistical interactions: using <inline-formula id="inf32">
<mml:math id="m37">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>_</mml:mo>
<mml:mo>&#x2061;</mml:mo>
<mml:mi mathvariant="italic">max</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> causes NPDR and Relief-based methods to become myopic; that is, focused on the importance of single variables (<xref ref-type="bibr" rid="B15">McKinney et al., 2013</xref>; <xref ref-type="bibr" rid="B7">Dawkins and McKinney, 2025</xref>). Once an appropriate neighborhood is determined, the imbalance between hit and miss groups in the neighborhood of the sample is accounted for by regression model weights using the ratio 1 - <italic>num_in_class</italic>/<italic>num_samples</italic>.</p>
<p>The nearest neighbors are determined from a chosen distance metric in the full space of variables. For the current study, we employ a novel distance metric called the Unsupervised Random Forest Proximity (URFP), chosen due to its ability to account for a non-isotropic variable space and its performance in the biosignature dataset compared with a traditional Manhattan distance metric (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>).</p>
</sec>
<sec id="s2-2">
<title>2.2 Random forest variable importance</title>
<p>We compare local-NPDR with other local importance scores, including local-RF, a method native to the RF classifier (<xref ref-type="bibr" rid="B2">Breiman, 2001</xref>). In global-RF variable permutation importance, the oob samples are fed into each tree of the forest to compute classification accuracy. By definition, the oob samples (about one-third of the training samples) are not seen by a particular tree during training (and varies depending on the tree in the forest). This accuracy calculation is repeated, but the order of the values for each variable is permuted in separate iterations. The change in average classification accuracy before and after permuting the variable is a measure of the variable&#x2019;s importance. Because permutation of an important variable is expected to decrease classification accuracy, the greater the decrease in accuracy after permutation, the more important the variable is considered (globally) for prediction. The local-RF variable importance procedure also computes changes in accuracy before and after variable permutation. However, instead of permuting the variable for all oob samples, the value of each variable is permuted for a <italic>single sample</italic>. That sample is then run through all trees in the forest <italic>for which it is oob</italic> to yield an average accuracy before and after variable permutation. The difference in accuracy is the local RF variable importance for that sample.</p>
</sec>
<sec id="s2-3">
<title>2.3 Procedure for reporting false predictions</title>
<p>We use NPDR to determine local and global feature importance. Because NPDR is a contrastive method, it predicts the class difference of neighbors, not the class of a given sample. To predict the class of individual samples, we use RF classification because of its robustness to skewed variables and mixed data types and its resistance to over-fitting. For each sample, we compute the local-NPDR importance scores for the features that were selected globally by LASSO-NPDR using the URFP distance metric; these are the variables on which the RF classifier is trained, ensuring that the feature selection method is independent from the classification method. In this step, a new URFP distance metric using only the global-NPDR features is used. The local-NPDR variable importance scores can be concordant with the global-NPDR scores (manifested as positive variable importance scores) or discordant (negative importance scores). If the sum of the local scores is negative (overall discordant), the sample was likely classified based on variables that were not part of the general (global) pattern of the classifier. We hypothesize that such samples are more likely to be false predictions because they do not follow the general pattern learned by the classifier from the global dataset. We combine local-NPDR feature importance scores with RF prediction probabilities to further constrain which samples are identified as potential false predictions, hypothesizing that samples classified with lower prediction probabilities are more likely to be incorrectly classified.</p>
<p>We further compare false prediction diagnosis of individual samples in holdout data using local-NPDR variable importance with local-RF. We compute the overall local variable importance scores for correctly and incorrectly classified samples (based on the RF classifier) to see whether discordance is associated with false predictions. An initial question is which globally important variables to include in the concordance calculation. NPDR can use a LASSO penalty that results in a statistical threshold for importance. However, RF does not have a threshold for feature selection. Thus, we use the global-NPDR features as the RF model variables to determine local-RF feature importance. NPDR feature selection thresholds can be defined either through P-values or via regularization (<xref ref-type="bibr" rid="B26">Tibshirani, 1996</xref>; <xref ref-type="bibr" rid="B28">Zou and Hastie, 2005</xref>).</p>
</sec>
<sec id="s2-4">
<title>2.4 Validating local-NPDR: real and simulated datasets</title>
<p>We validate the local-NPDR variable importance method on both real and simulated datasets. Simulated datasets allow us to compare effects of variable correlation, main and interaction effects, and class imbalance on ML models. Furthermore, since it is known whether variables in the simulated datasets are functional (<italic>i.e.</italic>, whether they have a main and/or an interaction effect), we can quantify the performance of our methods. Real datasets ensure that our methods work in real applications on imperfect or complex data. The real and simulated datasets used in this study are summarized in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap id="T1" position="float">
<label>TABLE 1</label>
<caption>
<p>Summary of the real biosignature data and three simulated datasets (top). For the simulated data, the number and type of functional features are given (main effects, interaction effects or noise features). Random Forest is used for training and testing accuracies for all data (additional accuracy information in <xref ref-type="fig" rid="F3">Figure 3</xref>) using global-NPDR-LURF feature selection (features listed at bottom). Biosignature features include IRMS and time-series derived features. Simulated data include functional features, which begin with &#x201c;main&#x201d; and &#x201c;int&#x201d; for main effects and interactions, respectively. Simulated features that begin with &#x201c;var&#x201d; are noise variables not involved in classification except by chance. Dashes are used for data that have fewer important features selected.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="left"/>
<th align="left">Biosignature data</th>
<th align="center">Simulation 1</th>
<th align="center">Simulation 2</th>
<th align="center">Simulation 3</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">class1: class0 (test)samples</td>
<td align="center">89:51 (22:12) abio:bio</td>
<td align="center">144:96 (36:24)</td>
<td align="center">144:96 (36:24)</td>
<td align="center">200:200 (50:50)</td>
</tr>
<tr>
<td align="center">main: interact: noise</td>
<td align="center">104 total features</td>
<td align="center">10:10:80 features</td>
<td align="center">10:10:80 features</td>
<td align="center">10:10:80 features</td>
</tr>
<tr>
<td align="center">Strength: (main/interact)</td>
<td align="center">&#x2014;</td>
<td align="center">1.5/1.5</td>
<td align="center">1.5/1.5</td>
<td align="center">0.8/1.5</td>
</tr>
<tr>
<td align="center">Train (Test) Accuracy</td>
<td align="center">90.7% (91.2%)</td>
<td align="center">77.9% (71.1%)</td>
<td align="center">82.9% (78.3%)</td>
<td align="center">84.3% (80.0%)</td>
</tr>
<tr>
<td colspan="5" align="center">Global NPDR Features</td>
</tr>
<tr>
<td align="right">1</td>
<td align="center">avg_rR<sup>45</sup>CO<sub>2</sub>/<sup>44</sup>CO<sub>2</sub>
</td>
<td align="center">mainvar9</td>
<td align="center">mainvar8</td>
<td align="center">mainvar5</td>
</tr>
<tr>
<td align="right">2</td>
<td align="center">sd_<italic>&#x3b4;</italic>
<sup>18</sup>O/<italic>&#x3b4;</italic>
<sup>13</sup>C</td>
<td align="center">mainvar4</td>
<td align="center">intvar8</td>
<td align="center">mainvar9</td>
</tr>
<tr>
<td align="right">3</td>
<td align="center">diff2_acf1</td>
<td align="center">mainvar1</td>
<td align="center">mainvar9</td>
<td align="center">mainvar1</td>
</tr>
<tr>
<td align="right">4</td>
<td align="center">fluctuation</td>
<td align="center">intvar8</td>
<td align="center">mainvar7</td>
<td align="center">mainvar4</td>
</tr>
<tr>
<td align="right">5</td>
<td align="center">time_kl_shift</td>
<td align="center">intvar3</td>
<td align="center">var14</td>
<td align="center">mainvar7</td>
</tr>
<tr>
<td align="right">6</td>
<td align="center">&#x2014;</td>
<td align="center">mainvar8</td>
<td align="center">var64</td>
<td align="center">mainvar10</td>
</tr>
<tr>
<td align="right">7</td>
<td align="center">&#x2014;</td>
<td align="center">mainvar2</td>
<td align="center">intvar7</td>
<td align="center">mainvar3</td>
</tr>
<tr>
<td align="right">8</td>
<td align="center">&#x2014;</td>
<td align="center">var64</td>
<td align="center">var6</td>
<td align="center">mainvar8</td>
</tr>
<tr>
<td align="right">9</td>
<td align="center">&#x2014;</td>
<td align="center">var35</td>
<td align="center">&#x2014;</td>
<td align="center">intvar4</td>
</tr>
<tr>
<td align="right">10</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">&#x2014;</td>
<td align="center">mainvar6</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We perform RF classification and global-NPDR feature selection for all datasets using an 80:20 train:test split that preserves the class imbalance. Previously, we found 80:20 splits have very stable test accuracies across repeated 5 folds (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>). In the current study, we use a single 80:20 split of the data to simplify the interpretation of the results while comparing the local score methods. We choose a split of the real data with a typical (median) test accuracy. RF hyperparameters are tuned using 5-fold cross-validation in the training set. The real astrobiology dataset consists of isotope ratio mass spectrometry (IRMS) measurements of volatile CO<sub>2</sub> evolved from laboratory-generated ocean world (OW) analogue brines of <italic>biotic</italic> and <italic>abiotic</italic> samples (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>). This biosignature dataset, referred to as Benchmark Ocean Worlds-&#x3b4;CO<sub>2</sub> dataset (BOW-&#x3b4;CO<sub>2</sub>), contains 174 samples of IRMS experiments (111 <italic>abiotic</italic> and 63 <italic>biotic</italic>), generated with 0.3% CO<sub>2</sub> by volume and containing different salt compositions relevant for both Europa and Enceladus. The imbalance in this dataset is 0.64, with <italic>biotic</italic> samples making up the minority class.</p>
<p>We generate three simulated datasets (summarized at the top of <xref ref-type="table" rid="T1">Table 1</xref>) using the <monospace>createSimulation2</monospace> function from our <italic>npdr</italic> R library based on the approach in Ref. (<xref ref-type="bibr" rid="B9">Lareau et al., 2015</xref>). These simulated data have the advantage of having known ground truth functional features (<italic>i.e.</italic>, features associated with the outcome variable) while incorporating realistic effects found in real data. Simulations 1 and 2 datasets are designed to have similar properties to the real BOW-&#x3b4;CO<sub>2</sub> dataset. Like our real dataset, these two simulated datasets have a similar number of features (<italic>p</italic> &#x3d; 100), sample size (<italic>m</italic> &#x3d; 240 train and 60 test samples) and class imbalance (0.6). In addition, the simulated data have a realistic correlation structure between variables and includes both interaction and main effects. We simulated 20% of the features to be functional, with 10 main effects (&#x201c;mainvars&#x201d;) and 10 interaction variables (&#x201c;intvars&#x201d;). These two simulations have effect sizes of 1.5 for both main effects and interaction effects. The remaining features are noise variables that have no effect on classification outcome. Since the main and interaction effects are known for particular variables, this allows us to assess whether feature selection methods are selecting relevant variables for classification. The Simulation 3 dataset has the same number of features (<italic>p</italic> &#x3d; 100) but a larger sample size (<italic>m</italic> &#x3d; 400 training samples and 100 test samples) with balanced classes instead of imbalance. This dataset has the same number of main effects and interactions as Simulations 1 and 2 but has smaller main effect sizes (main effect strength &#x3d; 0.8, interaction effect strength &#x3d; 1.5).</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec id="s3-1">
<title>3.1 Global-NPDR feature selection and RF classifiers</title>
<p>Before performing local-NPDR on individual samples, we first perform global-NPDR using all training samples and train RF classification models for the real biosignature data and the three simulated datasets. For global-NPDR feature selection, we use a LASSO penalty (see <xref ref-type="disp-formula" rid="e5">Equation 5</xref> in <xref ref-type="sec" rid="s2-1">Section 2.1</xref>) and URFP distance (global-NPDR-LURF) for training data splits. The dataset was split into train (89 <italic>abiotic</italic> and 22 <italic>biotic</italic>) and test (51 <italic>abiotic</italic> and 12 <italic>biotic</italic>) sets that reflect the global imbalance (see <xref ref-type="table" rid="T1">Table 1</xref> for further details). We train RF classifiers with tuned hyperparameters in the selected-feature spaces. For all datasets, we use weights to compensate for class imbalance in RF classifier training where <italic>class_weights &#x3d; 1/num_class</italic>. For the biosignature training data (<italic>m &#x3d;</italic> 140 samples, 89 <italic>abiotic</italic> and 51 <italic>biotic</italic>), global-NPDR with hyperparameter <inline-formula id="inf33">
<mml:math id="m38">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; 0.01 results in five selected features (bottom left of <xref ref-type="table" rid="T1">Table 1</xref>) out of 104 total predictors. Using these five features, the RF classifier with tuned hyperparameters <italic>mtry</italic> &#x3d; 5, <italic>splitrule</italic> &#x3d; &#x201c;extratrees&#x201d;, <italic>min. node.size</italic> &#x3d; 7, and <italic>ntrees</italic> &#x3d; 5,000 yields a training accuracy of 90.7% and a test accuracy of 91.2% (<xref ref-type="fig" rid="F3">Figure 3a</xref>). The accuracy breakdown by class (<italic>biotic</italic>/<italic>abiotic</italic>) shows high prediction accuracies for both classes in the train and test data alike for the biosignature dataset, despite the class imbalance. The <italic>abiotic</italic> class accuracy in the training data is 91.0% and in the test data it is 95.5%. For the <italic>biotic</italic> class, in the training data the RF prediction accuracy using NPDR-LURF features is 90.2% and in the test data it is 83.3%, slightly lower.</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>Random Forest (RF) train and test accuracies for real biosignature data and simulated datasets. <bold>(a)</bold> The RF classifier for biosignatures yields a 90.7% training accuracy. There are 51 <italic>biotic</italic> samples and 89 <italic>abiotic</italic> samples in the training data; five <italic>biotic</italic> samples are misclassified as <italic>abiotic</italic> (false negatives) and eight <italic>abiotic</italic> sample are predicted to be <italic>biotic</italic> (false positives). The biosignature train and test data show a similar high-accuracy performance despite class imbalance (class imbalance &#x3d; 0.64), where the overall test accuracy is 91.2%. <bold>(b)</bold> Simulation 1 is an imbalanced simulated dataset (class imbalance &#x3d; 0.6) that contains 144 class-<italic>1</italic> training samples and 96 class-<italic>0</italic> training samples, yielding an overall training accuracy of 77.9%. This dataset shows a more balanced class accuracy in the training data than the testing data. <bold>(c)</bold> Simulation 2 data is also imbalanced (class imbalance &#x3d; 0.6) with the same The two imbalanced simulated datasets show a discrepancyclass breakdown as Simulation 1 and shows similar behavior in terms of class accuracy imbalance in the test data but is even more pronounced. <bold>(d)</bold> Simulation 3 data is balanced and has a higher sample size (training data contains 200 class-<italic>0</italic> and class-<italic>1</italic> samples each). The class accuracies are balanced in both train and test data.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g003.tif">
<alt-text content-type="machine-generated">Four confusion matrices labeled a to d display classification data. a. Biosignature Data: Shows train accuracy of 90.7% and test accuracy of 91.2%. Class accuracies are 91.0% for Abiotic and 90.2% for Biotic in training; 95.5% for Abiotic and 83.3% for Biotic in testing.b. Simulation 1: Shows train accuracy of 77.9% and test accuracy of 71.7%. Class accuracies are 81.3% for Class 1 and 75.7% for Class 0 in training; 63.9% for Class 1 and 83.3% for Class 0 in testing.c. Simulation 2: Shows train accuracy of 82.9% and test accuracy of 78.3%. Class accuracies are 72.9% for Class 1 and 97.9% for Class 0 in training; 66.7% for Class 1 and 95.8% for Class 0 in testing.d. Simulation 3: Shows train accuracy of 84.3% and test accuracy of 80.0%. Class accuracies are 84.0% for Class 1 and 84.5% for Class 0 in training; 78.0% for Class 1 and 82.0% for Class 0 in testing.</alt-text>
</graphic>
</fig>
<p>For datasets Simulations 1-3, global-NPDR selected nine, eight, and ten features out of 100 (bottom of <xref ref-type="table" rid="T1">Table 1</xref>) with hyperparameter <inline-formula id="inf34">
<mml:math id="m39">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> &#x3d; {0.02, 0.013, and 0.01} respectively. The simulated datasets include functional features, which begin with &#x201c;main&#x201d; and &#x201c;int&#x201d; for main effects and interactions, respectively. Simulated features that begin with &#x201c;var&#x201d; are noise features and are not functional. The two simulated datasets (Simulations 1 and 2) that contain noise variables are also imbalanced datasets. The Simulation 3 dataset has balanced classes, and global-NPDR found no noise variables. This lower false positive rate for functional variables may be due to class balance or larger sample size. The resulting RF training (test) accuracies for data Simulations 1-3 are 77.9% (71.7%), 82.9% (78.3%), and 84.3% (80.0%), respectively (<xref ref-type="fig" rid="F3">Figures 3b&#x2013;d</xref>). The dataset with the highest accuracy (<xref ref-type="fig" rid="F3">Figure 3d</xref>) is balanced between classes and has a higher sample size. In addition, main effects play a more prominent role in feature selection (<xref ref-type="table" rid="T1">Table 1</xref>, bottom last column). For the respective simulated data, the tuned RF hyperparameters were: <italic>mtry</italic> &#x3d; {5, 8, 2}, <italic>splitrule</italic> &#x3d; {&#x201c;gini&#x201d;, &#x201c;extratrees&#x201d;, &#x201c;extratrees&#x201d;}, <italic>min. node.size</italic> &#x3d; {12, 3, 7}, and <italic>ntrees</italic> &#x3d; {5,000, 6,000, 6,000}. The two imbalanced simulated datasets show a discrepancy in class accuracy that is most notable in the test data (<xref ref-type="fig" rid="F3">Figures 3b,c</xref>), while the balanced simulated dataset shows a more balanced RF class prediction accuracy in both the train and test data (<xref ref-type="fig" rid="F3">Figure 3d</xref>).</p>
</sec>
<sec id="s3-2">
<title>3.2 Local-NPDR feature importance for true and false ML predictions</title>
<p>In the following sections we present the results of our local-NPDR feature importance method to discriminate between true and false ML predictions in the three simulated datasets as well as the real biosignature BOW-&#x3b4;CO<sub>2</sub> dataset.</p>
<sec id="s3-2-1">
<title>3.2.1 Local-NPDR feature importance for simulated data</title>
<p>For each sample in the train and test data, we calculate local-NPDR feature importance scores using a Ridge penalty, &#x201c;lambda.1se&#x201d; hyperparameter, and URFP distance for the set of global-NPDR-LURF features. Results from the test data are discussed here; see <xref ref-type="sec" rid="s12">Supplementary Section S3</xref> for training data results. For each sample, the total local-NPDR variable importance scores are computed (for the globally important features), and we use a t-test to compare the total local scores (TLS) between samples with a true and false prediction by the RF model. For the three simulated datasets <xref ref-type="table" rid="T1">Table 1</xref>, the total local-NPDR scores are higher in the true versus false prediction groups for both training (<xref ref-type="sec" rid="s12">Supplementary Figure S3</xref>) and test samples (<xref ref-type="fig" rid="F4">Figure 4</xref>). The elevated TLS in true versus false groups in the test data is statistically significant (with P-values: <inline-formula id="inf35">
<mml:math id="m40">
<mml:mrow>
<mml:mn>4.6</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf36">
<mml:math id="m41">
<mml:mrow>
<mml:mfenced open="" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mn>7.9</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>12</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>Total local-NPDR variable importance scores for true and false predictions in three simulated test (holdout) datasets. Total Local Scores (TLS) are computed for the globally important features based on LASSO NDPR. In each dataset, the local-NPDR scores are higher in the true (blue) versus false (red) prediction samples with very low overlap (all t-tests statistically significant). Detailed properties of Simulation 1-3 are given in <xref ref-type="table" rid="T1">Table 1</xref>. <bold>(a)</bold> Simulation 1 is class-imbalanced and the mean total local-NPDR feature importance scores are higher in the true prediction group (P &#x3d; <inline-formula id="inf42">
<mml:math id="m47">
<mml:mrow>
<mml:mn>4.6</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>). <bold>(b)</bold> Simulation 2 data is also imbalance and the mean local-NPDR variable importance scores is higher in the true prediction group (<inline-formula id="inf43">
<mml:math id="m48">
<mml:mrow>
<mml:mfenced open="" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>7.9</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>12</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. <bold>(c)</bold> Simulation 3 has a larger sample size, and the classes are balanced. Local-NPDR importance scores are also higher for true predictions <inline-formula id="inf44">
<mml:math id="m49">
<mml:mrow>
<mml:mfenced open="(" close="" separators="|">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2.1</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>12</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>).</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g004.tif">
<alt-text content-type="machine-generated">Three violin plots labeled a, b, and c, depict the total local-NPDR values under different simulations. Simulation 1 (43 TP/TN, 17 FP/FN samples) shows a P-value of 4.6 x 10^-6; Simulation 2 (4 TP/TN, 13 FP/FN samples) has a P-value of 7.9 x 10^-12; Simulation 3 (80 TP/TN, 20 FP/FN samples) indicates a P-value of 2.1 x 10^-12. Blue represents TP/TN and red represents FP/FN results.</alt-text>
</graphic>
</fig>
<p>True and false predictions can be further broken down into true positive/true negatives and false positives/false negatives. For the simulated datasets, class-<italic>0</italic> is taken to be the positive class. For the imbalanced datasets, this corresponds to the minority class, chosen to mimic the study design for the biosignature data. This means the true positive (TP) and false negative (FN) predictions involve the minority class<italic>-0</italic> for the two imbalanced simulated datasets, while true negative (TN) and false positive (FP) predictions involve samples of majority class<italic>-1</italic>.</p>
<p>Mean total local-NPDR variable importance scores for the imbalanced simulated datasets show different values for false negative versus false positive predictions in both the train (<xref ref-type="sec" rid="s12">Supplementary Figure S4</xref>) and test data (<xref ref-type="fig" rid="F5">Figures 5a,b</xref>). In both cases, the FP group, composed of class-<italic>1</italic> samples incorrectly predicted to be class-<italic>0,</italic> has higher mean TLS than the FN group, made of class-<italic>0</italic> samples incorrectly predicted to be class-<italic>1</italic>. This effect is likely due to class imbalance and is absent in the balanced simulated dataset (compare <xref ref-type="fig" rid="F5">Figures 5a,b</xref> with <xref ref-type="fig" rid="F5">Figure 5c</xref>), suggesting that local importance methods may be less reliable in the presence of imbalance, which is also a perennial challenge for classification methods. During RF classifier training of imbalanced datasets, the majority class may be penalized, and the algorithm attempts to maximize the classification accuracy of the minority class. This results in a higher classification error for the majority class.</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Mean total local-NPDR variable importance scores for three simulated datasets [left panel, <bold>(a&#x2013;c)</bold>] and mean RF prediction probability for the same three simulated datasets [right panel, <bold>(d&#x2013;f)</bold>]. Values are broken down by prediction type on the x-axes (FN &#x3d; false negative, FP &#x3d; false positive, TN &#x3d; true negative, TP &#x3d; true positive). False predictions are indicated by red circles, true predictions by blue, and the number of samples of each prediction type are given next to the points. Class sample sizes are indicated in the legend for each dataset. Each row of figures is one of the three simulated datasets (accuracies summarized in <xref ref-type="fig" rid="F3">Figures 3b&#x2013;d</xref>). <bold>(a)</bold> The distribution of mean total local-NPDR variable importance scores by prediction type is affected by the class imbalance. The mean total local-NPDR variable importance score is high for true positives (class-<italic>0</italic> samples correctly predicted to be class-<italic>0</italic>), true negative samples (class-<italic>1</italic> samples correctly classified), and false positives (class-<italic>1</italic> samples incorrectly predicted to be class-<italic>0</italic>). <bold>(b)</bold> The mean total local-NPDR feature score distribution by prediction type is similarly distributed to those in <bold>(a)</bold> (compare simulation parameters). <bold>(c)</bold> The mean total local-NPDR variable importance scores show a different distribution. In this case, the classes are balanced, there are more samples, and the main effect size is decreased to 0.8, while the interaction effects are kept at 1.5. For this dataset, the false negative and false positive scores are both lower than the true negative and true positive scores. <bold>(d)</bold> The mean RF prediction probability for the same simulated test samples as in <bold>(a)</bold> shows lowest prediction probability for false negatives, followed by false positives. <bold>(e)</bold> False positive and false negative mean RF prediction probabilities are lower than for true predictions for the same dataset as in <bold>(b)</bold>. <bold>(f)</bold> For the simulated samples in <bold>(c)</bold>, mean RF prediction probabilities are again lower for false predictions than for true predictions.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g005.tif">
<alt-text content-type="machine-generated">Six scatter plots labeled a to f show results from a random forest classifier. Plots a, b, and c display mean total local-NPDR importance scores, with respective test accuracies of 71.7%, 78.3%, and 80.0%. Plots d, e, and f show mean RF prediction probabilities for the same datasets. Red dots indicate false predictions and blue dots true predictions, with sample sizes of Class 0 and Class 1 indicated. Each plot is categorized by true positive (TP), true negative (TN), false positive (FP), and false negative (FN) outcomes.</alt-text>
</graphic>
</fig>
<p>However, the RF prediction probabilities can help differentiate the true and false predictions in these cases (<xref ref-type="fig" rid="F5">Figures 5d,e</xref>), where the average probabilities for false predictions are lower than those for true predictions. For the imbalanced datasets, the mean RF prediction probability for true positives (&#x223c;70%), representing samples of the minority class, is lower than for true negatives (&#x3e;90%), samples of the majority class.</p>
<p>The balanced simulated dataset is less prone to the discrepancies in false prediction mean total local-NPDR variable importance scores (<xref ref-type="fig" rid="F5">Figure 5c</xref>). In this case, both the FP and FN predictions have similarly low mean total local-NPDR importance scores and both the TP and TN predictions have much higher scores. For this dataset, the average RF prediction probabilities are also more balanced among prediction types, with false prediction probabilities both appearing at &#x223c;65% or below and mean true prediction probabilities both being &#x223c;75% (<xref ref-type="fig" rid="F5">Figure 5e</xref>). This average probability being lower than the probability for the majority class in the imbalanced simulated datasets could be related to the lower sample size. However, because of the potential for TLS overlap between individual true and false predictions, especially in imbalanced datasets, it can be beneficial to incorporate other information for diagnosing sample predictions, such as the RF classifier probability.</p>
</sec>
<sec id="s3-2-2">
<title>3.2.2 Local-NPDR feature importance for biosignature data</title>
<p>For the biosignature data, the <italic>biotic</italic> class corresponds to the positive class. This means that TP and FN predictions involve the minority <italic>biotic</italic> class, while TN and FP predictions involve samples of the majority <italic>abiotic</italic> class. Since the sample sizes are small for the biosignature test data (for example, one prediction type, FP, has only one sample), we will discuss the mean total local-NPDR variable importance scores for the training data. In <xref ref-type="sec" rid="s3-3">Section 3.3</xref>, we present results using local-NPDR variable importance in combination with RF prediction probabilities to diagnose false predictions on holdout simulated and biosignature data.</p>
<p>For the biosignature training data, both the FP and FN mean total local-NPDR importance score is higher than the TN, representing the <italic>abiotic</italic> samples (<xref ref-type="fig" rid="F6">Figure 6a</xref>). This could be due to the much smaller sample size. If more samples were added, we might expect a distribution that more resembles that of the local-NPDR mean TLS in the imbalanced simulated datasets. Like the imbalanced simulated data, the mean RF prediction probabilities for the false predictions in the biosignature data are much lower than those for the true predictions (<xref ref-type="fig" rid="F6">Figure 6b</xref>). The combination of indicators provided by both the mean total local-NPDR variable importance scores and the RF prediction probabilities provide a complimentary approach for identifying possible false or problematic predictions (in which the model is unsure of classification) in samples whose actual class is unknown.</p>
<fig id="F6" position="float">
<label>FIGURE 6</label>
<caption>
<p>Mean total local-NPDR variable importance and mean RF prediction probability for the four different prediction types in the biosignature training data. The number of samples representing each type of prediction are indicated next to the points. <bold>(a)</bold> Local-NPDR mean total local importance score for each prediction type for training samples in the RF biosignature classification model. The dataset has a class imbalance of 0.64, where <italic>biotic</italic> is the minority class (and is also designated the positive class). This dataset shows a mean total local-NPDR importance score for FP samples that is larger than the score for TN samples, which mirrors behavior seen in the imbalanced simulated data (compare a and <xref ref-type="fig" rid="F4">Figures 4a,b</xref>). <bold>(b)</bold> The mean RF prediction probability is below 70% for both FN and FP samples and higher for TN (&#x3e;85%) and TP (&#x3e;77%) samples.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g006.tif">
<alt-text content-type="machine-generated">Comparison between Local-NPDR importance score and Random Forest prediction probability for a dataset. Panel a shows Local-NPDR scores with false predictions in red and true predictions in blue. Panel b illustrates Random Forest probabilities with the same false and true prediction color scheme. The random forest model has a training accuracy of 90.7 percent and a class imbalance of 0.64, with the positive class being biotic. Data points are labeled for false negatives, false positives, true negatives, and true positives.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s3-2-3">
<title>3.2.3 Comparison of local feature importance methods</title>
<p>Using our RF models trained in the global-NPDR-LURF feature space as the base classifier for each dataset, we compare local-NPDR with local-RF feature importance. Details for each algorithm can be found in the Methods section. As with local-NPDR, we perform a t-test for the local-RF TLS between true and false predictions. For the three simulated datasets, the total local-RF scores are higher in the true versus false prediction groups for both training (<xref ref-type="sec" rid="s12">Supplementary Figure S3</xref>) and test samples (<xref ref-type="fig" rid="F7">Figure 7</xref>). The elevated TLS in true versus false groups in the test data is statistically significant with P-values ranging from <inline-formula id="inf37">
<mml:math id="m42">
<mml:mrow>
<mml:mn>1.6</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> to <inline-formula id="inf38">
<mml:math id="m43">
<mml:mrow>
<mml:mn>2.2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>16</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. Again, there is the potential for score overlap between individual true and false predictions, meaning that information provided by the RF probability model could be useful in identifying false predictions.</p>
<fig id="F7" position="float">
<label>FIGURE 7</label>
<caption>
<p>Total local-RF variable importance scores for true and false predictions in three simulated test (holdout) datasets. Total Local Scores (TLS) are computed for the globally important features based on LASSO NDPR. In each dataset, the local-RF scores are higher in the true (blue) versus false (red) prediction samples (all t-tests statistically significant). Detailed properties of Simulation 1-3 are given in <xref ref-type="table" rid="T1">Table 1</xref>. <bold>(a)</bold> Simulation 1 is class-imbalanced has and the mean total local-RF importance scores that are higher in the true prediction group (P &#x3d; <inline-formula id="inf39">
<mml:math id="m44">
<mml:mrow>
<mml:mn>1.6</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>6</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>). <bold>(b)</bold> Simulation 3 is also class imbalanced and the mean local-RF variable importance scores are higher in the true prediction group (<inline-formula id="inf40">
<mml:math id="m45">
<mml:mrow>
<mml:mfenced open="" close=")" separators="|">
<mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3c;</mml:mo>
<mml:mn>2.2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>16</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>. <bold>(c)</bold> Local-RF importance scores are in the balanced dataset Simulation 3 are higher for true predictions <inline-formula id="inf41">
<mml:math id="m46">
<mml:mrow>
<mml:mfenced open="(" close="" separators="|">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2.0</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mn>10</mml:mn>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>10</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</inline-formula>).</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g007.tif">
<alt-text content-type="machine-generated">Three violin plots illustrating simulation results. a. Simulation 1: Displays total local-RF for TP/TN (blue) and FP/FN (red) with 43 and 17 samples, respectively. P-value is 1.6 &#xD7; 10^-6.b. Simulation 2: Displays total local-RF for TP/TN (blue) and FP/FN (red) with 47 and 13 samples, respectively. P-value is 2.2 &#xD7; 10^-16.c. Simulation 3: Displays total local-RF for TP/TN (blue) and FP/FN (red) with 80 and 20 samples, respectively. P-value is 2.0 &#xD7; 10^-10.</alt-text>
</graphic>
</fig>
<p>An analysis of the mean TLS for local-RF for train and test samples in the three simulated and biosignature datasets versus prediction type (FP, FN, TP, and TN) shows separation between both classes of false predictions and true predictions (<xref ref-type="fig" rid="F8">Figure 8</xref>; <xref ref-type="sec" rid="s12">Supplementary Figure S4</xref>). While mean TLS for local-NPDR in some false predictions are higher than mean TLS for some true predictions, local-RF mean-TLS for false predictions are always lower than mean-TLS for true predictions (compare <xref ref-type="fig" rid="F5">Figures 5</xref>, <xref ref-type="fig" rid="F8">8</xref>). Local-RF variable importance is expected to have a good performance at identifying false predictions, since this method is native to the classifier, and these results show that local-RF importance is less affected by class imbalance than local-NPDR. Limitations to local-RF variable importance were mentioned in <xref ref-type="sec" rid="s1">Section 1</xref> are discussed more in <xref ref-type="sec" rid="s4">Section 4</xref>.</p>
<fig id="F8" position="float">
<label>FIGURE 8</label>
<caption>
<p>Mean total local score (TLS) for local-RF variable importance for four datasets. The number of samples representing each type of prediction are indicated next to the points. <bold>(a)</bold> Local-RF mean TLS for each prediction type for test samples in the simulated dataset with 71.7% RF test accuracy. Both FP and FN predictions have lower scores than TN and TP predictions. <bold>(b)</bold> Local-RF mean TLS for each prediction type for test samples in the simulated dataset with 78.3% RF test accuracy. Again, both false prediction groups show lower mean TLS than true predictions. <bold>(c)</bold> Mean TLS using local-RF for the simulated dataset with 80.0% RF test accuracy shows the most similar distribution between the two false prediction groups and the two true prediction groups separately, likely driven by class balance. <bold>(d)</bold> The mean local-RF TLS for the biosignature data training samples (depicted for increased sample sizes) shows lower FN and FP scores than TN and TP scores. Although imbalanced, this dataset shows the most similar TLS distribution to the balanced simulated dataset in <bold>(c)</bold>.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g008.tif">
<alt-text content-type="machine-generated">Four-panel chart showing mean total local-RF importance scores. Panels (a) and (b) compare RF test accuracy with imbalances and effects settings. Panel (c) shows 80.0% accuracy and two equal class sizes. Panel (d) presents RF train accuracy at 90.7% with class biotic as the minority. True predictions are blue, false predictions are red, with varying false negative (FN), false positive (FP), true negative (TN), and true positive (TP) counts.</alt-text>
</graphic>
</fig>
<p>Both local-NPDR and local-RF result in statistically significant differences in mean-TLS between true and false predictions in the simulated datasets (<xref ref-type="fig" rid="F4">Figures 4</xref>, <xref ref-type="fig" rid="F7">7</xref>). For the biosignature data, local-NPDR variable importance indicates less clear separation between true and false prediction scores (compare <xref ref-type="fig" rid="F6">Figures 6a</xref>, <xref ref-type="fig" rid="F8">8d</xref>), indicating the sensitivity of the multivariate regression to both imbalance and small sample sizes, discussed in more detail in <xref ref-type="sec" rid="s4">Section 4</xref>. In the next section, we compare the ability of local-RF and local-NPDR to diagnose false predictions in &#x201c;unknown&#x201d; samples across the four datasets. While the results in this section may lead one to conclude that local-RF will always outperform local-NPDR, two additional sets of analyses on the test samples reveal a comparable performance between the two methods.</p>
<p>We perform LIME and treeSHAP explanation on the three simulations (<xref ref-type="table" rid="T1">Table 1</xref>) using the same tuned RF model restricted to LURF features that was used for local-NPDR and local-RF. We train the LIME and treeSHAP explainers on this data and test TLS scores on the holdout data for differences between true and false RF predictions (<xref ref-type="fig" rid="F9">Figure 9</xref>). The total LIME and tree SHAP total scores are higher in the true prediction aggregated group (TP and TN combined) than the false prediction groups (FP and FN combined). However, the differences are less significant than local-NPDR (<xref ref-type="fig" rid="F4">Figure 4</xref>) and local-RF (<xref ref-type="fig" rid="F7">Figure 7</xref>), with some of the LIME (<xref ref-type="fig" rid="F9">Figure 9b</xref>) and treeSHAP (<xref ref-type="fig" rid="F9">Figure 9f</xref>) true/false differences not being statistically significant. The treeSHAP TLS distributions in the true prediction groups are consistently bimodal. This is due to the TN scores being positive and TP scores being negative. The TP scores are consistently in the opposite of the desired direction.</p>
<fig id="F9" position="float">
<label>FIGURE 9</label>
<caption>
<p>Total local variable importance scores for true and false predictions in three simulated test (holdout) datasets for LIME <bold>(a&#x2013;c)</bold> and treeSHAP <bold>(d&#x2013;f)</bold>. Total Local Scores (TLS) are computed for the globally important features based on LASSO NDPR. The mean TLS are higher in the true (blue) versus false (red) prediction samples, but the P-values and distribution overlap are higher between true and false groups than local-NPDR and local-RF. Detailed properties of Simulation 1-3 are given in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g009.tif">
<alt-text content-type="machine-generated">Violin plots compare total LIME and SHAP scores for three simulations. a. Simulation 1: LIME scores, significant difference (P &#x3d; 0.0048) between TP/TN (43 samples) and FP/FN (17 samples). b. Simulation 2: LIME scores, no significant difference (P &#x3d; 0.600) between TP/TN (47 samples) and FP/FN (13 samples). c. Simulation 3: LIME scores, significant difference (P &#x3d; 2.03 x 10^-5) between TP/TN (80 samples) and FP/FN (20 samples). d. SHAP scores, significant difference (P &#x3d; 0.0003) between TP/TN and FP/FN. e. SHAP scores, significant difference (P &#x3d; 0.0005) between TP/TN and FP/FN.f. SHAP scores, no significant difference (P &#x3d; 0.619) between TP/TN and FP/FN.</alt-text>
</graphic>
</fig>
</sec>
</sec>
<sec id="s3-3">
<title>3.3 Diagnosing false predictions in &#x201c;unknown&#x201d; samples</title>
<p>Four samples, each representing the four prediction types, are chosen for further analysis from test samples in each of the simulated datasets and the biosignature data. Local-NPDR and local-RF total local importance scores are calculated for each of the four samples, as well as the prediction probabilities; for each method, we attempt to characterize the results as either indicative of a true prediction or a false one. Results for the simulated datasets can be found in <xref ref-type="sec" rid="s12">Supplementary Section S4</xref>, and we present the results of this analysis for the biosignature dataset here.</p>
<p>Consider the case where the actual class of the four test samples is unknown. Local importance methods and classification probabilities can help us identify potentially false predictions in this scenario by analyzing the variable importance TLS for the samples, the individual local feature importance scores, and the RF classifier prediction probability (<xref ref-type="fig" rid="F10">Figure 10</xref>). For example, given the fact that an &#x201c;unknown&#x201d; sample has a positive total local-NPDR score of 19.4, that only one of the features has a negative local importance score, and that the classifier reports an 82.1% probability that the sample is <italic>biotic</italic>, we accept this prediction as a likely true positive (<xref ref-type="fig" rid="F10">Figure 10a</xref>). Likewise, for an <italic>abiotic</italic> classification with a high-magnitude positive local-NPDR score of 100.6, small-magnitude negative local scores for individual variables, and a RF prediction probability of 81.3%, we accept this classification as a likely true negative (<xref ref-type="fig" rid="F10">Figure 10b</xref>). If the TLS for local-NPDR is close to zero or negative, if there are large-magnitude negative local variable importance scores, and if the RF prediction probability is low, these samples are subject to being flagged as potential false predictions (<xref ref-type="fig" rid="F10">Figures 10c,d</xref>). For a sample with a local-NPDR score of &#x2212;33.35, a large-magnitude negative score for the top global-NPDR ranked feature and an RF prediction probability of 55.5%, this <italic>biotic</italic> prediction is flagged as a potential false positive (<xref ref-type="fig" rid="F10">Figure 10c</xref>). For an <italic>abiotic</italic> prediction with a similar large-magnitude negative score for the top global-NPDR feature, a TLS of &#x2212;67.6, and an RF classification probability of 63.3%, this sample is flagged as a likely false negative (<xref ref-type="fig" rid="F10">Figure 10d</xref>).</p>
<fig id="F10" position="float">
<label>FIGURE 10</label>
<caption>
<p>Typical distributions of local-NPDR variable importance scores for true and false predictions. Variables shown are listed according to global-NPDR importance ranking, with the most important feature for classification at the top. &#x201c;Contradicts&#x201d; (red) means the local-NPDR score of a variable is negative, in disagreement with the global-NPDR importance. &#x201c;Supports&#x201d; (blue) means the local-NPDR score for a variable is positive, in agreement with the global-NPDR importance. <bold>(a)</bold> For a <italic>biotic</italic> sample correctly classified as <italic>biotic</italic> by the RF biosignature model, a true positive, the overall local variable importance score is 19.4. This positive score indicates the local variable importance scores are mostly concordant with the assigned class for this sample (<italic>biotic</italic>). Additionally, the RF probability model reports a high prediction probability of 82.1%, increasing the likelihood that this is a correct prediction. <bold>(b)</bold> For an <italic>abiotic</italic> sample correctly predicted to be <italic>abiotic</italic>, a true negative, the total local score (TLS) is 100.6 and the RF prediction probability is 81.3%, suggesting this is a correct prediction. <bold>(c)</bold> For an <italic>abiotic</italic> sample incorrectly predicted to be <italic>biotic</italic>, a false positive, the TLS is &#x2212;33.4. This large negative score flags the sample as a potential false prediction in which the assigned classed, <italic>biotic</italic>, is not concordant with the local variable importance scores of the two most important global-NPDR features. In this case the RF probability model yields a low prediction probability of 55.5%, further increasing doubt in the validity of this classification. <bold>(d)</bold> For a <italic>biotic</italic> sample incorrectly predicted to be <italic>abiotic</italic>, a false negative, the local-NPDR TLS is &#x2212;67.6, with a large-magnitude negative score for the top global-NPDR feature. The RF probability is reported as 63.3%, and this along with the large negative TLS, indicates the sample is likely incorrectly classified.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g010.tif">
<alt-text content-type="machine-generated">Four bar charts labeled a to d, illustrating classification outcomes with local importance scores. Chart a shows a true positive with a score of 19.39. Chart b displays a true negative with a score of 100.63. Chart c depicts a false positive with a score of -33.35. Chart d shows a false negative with a score of -67.56. Bars are colored to indicate support (blue) or contradiction (red) for each outcome.</alt-text>
</graphic>
</fig>
<p>An analogous analysis using local-RF as the importance method for the same four test samples uses similar reasoning (<xref ref-type="fig" rid="F11">Figure 11</xref>). The magnitude of local-RF and local-NPDR variable importance scores differs because the nature of the methods is fundamentally different; local-RF importance scores are changes in accuracy after and before variable permutation, while local-NPDR scores represent regression coefficients for the pairwise sample regression. This means the magnitudes for the local-NPDR scores will vary by dataset, while the local-RF importance scores will always represent a change in percent accuracy. While the particular variables that are negative differ between local-RF and local-RF, the TLS for the true predictions are positive, in agreement with local-NPDR for the true positive (<xref ref-type="fig" rid="F11">Figure 11a</xref>) and true negative sample predictions (<xref ref-type="fig" rid="F11">Figure 11b</xref>). The TLS is negative for the false positive (<xref ref-type="fig" rid="F11">Figure 11c</xref>) and false negative (<xref ref-type="fig" rid="F11">Figure 11d</xref>) predictions, again in agreement with local-NPDR for these samples. In general, the concordance/discordance in true/false predictions is more pronounced in the local-RF scores for these samples than for the local-NPDR scores (compare the amount of blue in true and red in false predictions in <xref ref-type="fig" rid="F10">Figures 10</xref>, <xref ref-type="fig" rid="F11">11</xref>).</p>
<fig id="F11" position="float">
<label>FIGURE 11</label>
<caption>
<p>Local-RF variable importance scores for the cases analyzed by local-NPDR (<xref ref-type="fig" rid="F10">Figure 10</xref>). The variables are listed according to global-NPDR importance scores. &#x201c;Contradicts&#x201d; (red bars) means the local-RF score is negative (unimportant for classification) for a variable that is globally important, while &#x201c;Supports&#x201d; (blue bars) means the local variable importance score is positive, agreeing with the global importance of the variable. <bold>(a)</bold> A sample classified as <italic>biotic</italic> shows all positive local-RF importance scores in agreement with (supporting) the global variable importance scores. The total local score (TLS) is 0.18 and the RF prediction probability is 82.1%, indicating that this is likely a true classification of a biosignature. <bold>(b)</bold> With only one small negative variable importance score, this <italic>abiotic</italic> prediction has a TLS of 0.22 and a RF probability of 81.3%. We accept this abiotic classification as a likely true prediction. <bold>(c)</bold> A sample classified as <italic>biotic</italic> with a local-RF TLS of &#x2212;0.015 and a RF prediction probability of 55.5% is flagged as a potential false positive prediction. <bold>(d)</bold> An <italic>abiotic</italic> prediction with a TLS of &#x2212;0.07 and several negative variable importance scores has a 63.3% RF probability. This <italic>abiotic</italic> prediction is flagged as a potential false negative.</p>
</caption>
<graphic xlink:href="fspas-12-1651953-g011.tif">
<alt-text content-type="machine-generated">A set of four bar charts labeled a to d, illustrating classification outcomes and RF local variable importance scores. Chart a shows a true positive with a total local score of 0.176. Chart b shows a true negative with a total local score of 0.217. Chart c shows a false positive with a total local score of -0.015. Chart d shows a false negative with a total local score of -0.069. Each chart includes bars for different factors, with red indicating contradiction and blue indicating support for the classification.</alt-text>
</graphic>
</fig>
<p>To illustrate an applied comparative analysis, consider the case where we have a set of test samples with unknown actual classes and trained RF classifier and probability models. Suppose models are trained with global-NPDR selected features; each test sample is run through the classifier and probability RF models and is labeled with a prediction and probability. The goal of the local variable importance analysis is to &#x201c;quarantine&#x201d; the test samples that could be falsely predicted from the samples that we are most confident are correctly classified. To do this, we consider the class and probability output of the RF models in the context of the TLSs discussed above (<xref ref-type="fig" rid="F10">Figures 10</xref>, <xref ref-type="fig" rid="F11">11</xref>) and define an RF probability and an NPDR/RF TLS for which we accept samples. A challenge is to define an appropriate RF probability and TLS threshold to flag samples. A detailed discussion of potential strategies is deferred until the Discussion; for now, consider that the TLS threshold will be local importance method dependent (compare the differences in local-RF and local-NPDR importance scores) and subject to user preference, as will the probability threshold. For this analysis, we use an RF prediction probability threshold of 75% for all datasets and both local importance methods.</p>
<p>We use local-NPDR importance score thresholds of {0.25, 0.35, 0.35} for Simulations 1-3 with RF test accuracies of {71.7%, 78.3%, and 80.0%} respectively. These thresholds differ because of the different distributions of local-NPDR scores in the three datasets (see <xref ref-type="sec" rid="s12">Supplementary Figure S7</xref>). For the biosignature data, we use a local-NPDR score threshold of 25 (see <xref ref-type="sec" rid="s12">Supplementary Figure S8</xref>). Analysis of the distribution of local-RF importance scores for the training datasets show a similar distribution among the datasets due to the nature of the importance score calculation. We therefore define a threshold of 0 for local-RF importance for all datasets. False samples in the test data for each dataset can be detected using both methods (<xref ref-type="table" rid="T2">Table 2</xref>). For the arbitrary thresholds defined, local-RF and local-NPDR perform comparably well, showing similar or comparable overall false prediction diagnostic rates (compare third and fifth columns, <xref ref-type="table" rid="T2">Table 2</xref>). For the biosignature data, the two methods flag the same falsely predicted samples, one false positive and one false negative (compare biosignature data in columns 4 and 6), and the miss the same false negative.</p>
<table-wrap id="T2" position="float">
<label>TABLE 2</label>
<caption>
<p>False prediction diagnosis rates for local-RF and local-NPDR variable importance methods for four datasets using either a RF prediction probability &#x3c;75% or a TLS less than an arbitrary threshold.</p>
</caption>
<table>
<thead valign="top">
<tr>
<th align="center">Dataset</th>
<th align="center">RF test accuracy</th>
<th align="center">Local-RF false prediction diagnostic rate</th>
<th align="center">FP/FN diagnosis rate: local-RF</th>
<th align="center">Local-NPDR false prediction diagnostic rate</th>
<th align="center">FP/FN diagnosis rate: local-NPDR</th>
</tr>
</thead>
<tbody valign="top">
<tr>
<td align="center">Simulated 1</td>
<td align="center">71.70%</td>
<td align="center">52.90%</td>
<td align="center">61.5%/25.0%</td>
<td align="center">70.60%</td>
<td align="center">61.5%/100.0%</td>
</tr>
<tr>
<td align="center">Simulated 2</td>
<td align="center">78.30%</td>
<td align="center">76.90%</td>
<td align="center">75.0%/100.0%</td>
<td align="center">61.50%</td>
<td align="center">58.3%/100.0%</td>
</tr>
<tr>
<td align="center">Simulated 3</td>
<td align="center">80.00%</td>
<td align="center">75.00%</td>
<td align="center">72.7%/77.8%</td>
<td align="center">90.00%</td>
<td align="center">81.8%/100.0%</td>
</tr>
<tr>
<td align="center">Biosignature</td>
<td align="center">91.20%</td>
<td align="center">66.70%</td>
<td align="center">100.0%/50.0%</td>
<td align="center">66.70%</td>
<td align="center">100.0%/50.0%</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In this example analysis, local-RF quarantines fewer true predictions than local-NPDR in all datasets. For example, in the simulated dataset with 71.7% RF test accuracy, local-NPDR flags 26 total samples while local-RF only flags 12. Out of the 26 flagged by local-NPDR, 14 are true predictions with low TLS; out of the 12 flagged by local-RF, only two are true predictions. It is worth mentioning that global-NPDR feature selection has significantly enabled local-RF in this analysis; this will be discussed more in <xref ref-type="sec" rid="s4">Section 4</xref>. To summarize this applied analysis: local-RF and local-NPDR do a comparable job flagging false predictions in the three simulated datasets and the biosignature dataset. However, local-RF, when supplied with a model trained on global-NPDR selected features, flags significantly fewer true predictions than local-NPDR for all datasets.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>One minimal way of providing explainability for ML tools is to provide a set of variables or features that are important for predicting the outcome of interest (<xref ref-type="bibr" rid="B16">Montavon et al., 2018</xref>; <xref ref-type="bibr" rid="B24">Roscher et al., 2020</xref>), essentially feature selection (<italic>e.g.</italic>, global-NPDR). If ML model prediction is to provide a deeper level of explainability, analyses beyond global feature selection are needed. Previously, we provided tools for global feature selection, for network visualization of how those features work together to affect model predictions (<xref ref-type="bibr" rid="B9">Lareau et al., 2015</xref>; <xref ref-type="bibr" rid="B10">Le et al., 2020</xref>), and applied local-RF variable importance to assess the relative likelihood that an individual ML prediction is true or false (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>).</p>
<p>Here we present a new local feature importance tool, local-NPDR, by extending global-NPDR variable importance to compute importance scores in the neighborhood of a single sample. Local-NPDR uses a generalized linear model to contrastively determine whether neighbors of a sample of interest are in the same or different class (hits or misses). Variable importance scores are coefficients in the contrastive loss optimization, which can include LASSO or Ridge penalties. NPDR is sensitive to detecting interactions, a significant advantage in feature selection and importance ranking for high-dimensional datasets. We used an URFP distance metric to define the neighborhood because it effectively handles non-isotropic variable spaces and reduces correlation between variables (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>); however, NPDR can accept any number of different distance metrics. In addition to the distance metric, the choice of <italic>k</italic> affects NPDR&#x2019;s ability to detect interactions (<xref ref-type="bibr" rid="B7">Dawkins and McKinney, 2025</xref>). For global-NPDR, we used a default sample-size-dependent <italic>k</italic> that reliably balances interaction and main effects and accounts for class imbalance. For local-NPDR, we used the maximum number of neighbors to increase power to detect important variables.</p>
<p>We used the local-NPDR scores to explain RF predictions of individual samples, and we used the total local score (TLS) of the globally important variables to help diagnose false predictions. We showed that high positive TLS for samples is associated with true predictions, and low or negative TLS is associated with false predictions. The RF sample prediction probability provided additional true/false diagnostic evidence. Additional explainability of sample prediction can be provided by putting the local-NPDR feature importance scores in the context of main and interaction effects in a statistical interaction network. This interaction network (computed by the <monospace>regain</monospace> function in the NPDR R library) is a <italic>pxp</italic> matrix that contains each variable&#x2019;s main effect on the diagonal and interaction effects with other features on the off-diagonal entries (<xref ref-type="sec" rid="s12">Supplementary Figure S2</xref>). The centrality scores of the variables in this interaction network give the cumulative effects of the interactions and main effects of each variable (<xref ref-type="sec" rid="s12">Supplementary Table S1</xref>) (<xref ref-type="bibr" rid="B6">Davis et al., 2010</xref>; <xref ref-type="bibr" rid="B9">Lareau et al., 2015</xref>). The ability to go beyond an importance score and see the individual effects of each variable and interaction provides additional explainability.</p>
<p>The context of the interaction network and the local-NPDR score distributions for each prediction type can be used to understand which variables are important for each prediction type: FP, FN, TP, FN (<xref ref-type="sec" rid="s12">Supplementary Figure S3</xref>). This means the relative reliability of each feature can be understood and used to make a final decision about the likelihood of a particular sample being correctly or incorrectly predicted, providing another level of explanation with implications for astrobiology missions seeking isotopic biosignatures. For example, the top-ranked global-NPDR feature, <italic>avg</italic>_<italic>rR</italic>
<sup>
<italic>45</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub>
<italic>/</italic>
<sup>
<italic>44</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub> has a large interaction network centrality, with a moderate main effect while participating in two large-magnitude interactions (see <xref ref-type="sec" rid="s12">Supplementary Figure S2</xref>, node 1). From local-NPDR feature importance analysis on the biosignature training data, this variable is an important indicator for TPs and FPs, but a poor indicator of FNs and TNs (see <xref ref-type="sec" rid="s12">Supplementary Figure S2</xref>). This means for a sample labeled <italic>biotic</italic>, a large positive local-NPDR score for <italic>avg</italic>_<italic>rR</italic>
<sup>
<italic>45</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub>
<italic>/</italic>
<sup>
<italic>44</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub> indicates the prediction likely represents a true biosignature, and a negative or a low-magnitude score means a likely false biosignature prediction. However, this variable on average has negative scores for TN predictions while being on average positive for FNs. This contradictory behavior indicates that it is important to consider the interactions this variable participates in rather assume a large main effect is driving classification, since it is a good predictor for some biotic and abiotic samples but not all.</p>
<p>If a particular sample is predicted to be <italic>abiotic</italic>, the local-NPDR score for <italic>avg</italic>_<italic>rR</italic>
<sup>
<italic>45</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub>
<italic>/</italic>
<sup>
<italic>44</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub> should be considered in the context of other local variable importance scores. One of the largest statistical interactions that <italic>avg</italic>_<italic>rR</italic>
<sup>
<italic>45</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub>
<italic>/</italic>
<sup>
<italic>44</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub> participates in is with <italic>diff2_acf1</italic>, the third-ranked global NPDR feature. This feature is important for diagnosing FNs and TPs, and unimportant for FPs (<xref ref-type="sec" rid="s12">Supplementary Figure S2</xref>), meaning that the sign of <italic>diff2_acf1</italic> can help determine if a sample labeled <italic>abiotic</italic> is likely to be a FN or a TN&#x2013;if a sample labeled <italic>abiotic</italic> has a negative local-NPDR score for <italic>diff2_acf1</italic> and a large positive score for <italic>avg</italic>_<italic>rR</italic>
<sup>
<italic>45</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>
</sub>
<italic>/</italic>
<sup>
<italic>44</italic>
</sup>
<italic>CO</italic>
<sub>
<italic>2</italic>,</sub> it is likely that the prediction is false, despite the fact that the local score for the top-ranked global-NPDR feature is large and positive, and may dominate the TLS. This understanding may result in the acceptance of more true predictions and the rejection of more false predictions in our applied analysis in <xref ref-type="sec" rid="s3-3">Section 3.3</xref> if it were to be encoded in the algorithm to accept or quarantine individual samples. This analysis illustrates the complexity of an extremely small variable space in terms of variable interactions and main effects and how they work together to inform certain prediction types. Understanding and appreciating this nuance can enable increased explanation and scrutiny for individual sample predictions in biosignature classification.</p>
<p>We compared local-NPDR with widely-used local explainers: local-RF, LIME and treeSHAP. Local-NPDR and local-RF showed a statistically significant higher TLS for true versus false prediction samples for all simulated datasets (<xref ref-type="fig" rid="F4">Figures 4</xref> and <xref ref-type="fig" rid="F7">7</xref>). LIME and treeSHAP TLS were also higher in the true prediction samples, but there was more overlap between true and false distributions and some of the differences were not statistically significant (<xref ref-type="fig" rid="F9">Figure 9</xref>). Despite having the largest sample size and being class balanced, the treeSHAP TLS difference was not significant for Simulation 3 (<xref ref-type="fig" rid="F9">Figure 9f</xref>). This could be due to the stronger interaction effects in Simulation 3 compared to the main effects. Local-NPDR had the lowest P-value of its results for Simulation 3 (<xref ref-type="fig" rid="F4">Figure 4c</xref>) while this simulation did not yield the lowest P-value for local-RF (<xref ref-type="fig" rid="F7">Figure 7c</xref>). NPDR has been shown to have good power to detect interaction effects whereas RF has some limitations when detecting interactions. However, feature selection prior to running local-RF helps its sensitivity to interactions. The LIME TLS difference was not significant for Simulation 2 (<xref ref-type="fig" rid="F9">Figure 9b</xref>), which is imbalanced and has a lower sample size than Simulation 3.</p>
<p>The simulated data enabled us to compare the differential effects of class imbalance, sample size, and the relative magnitudes of main/interaction effects for variables, and we can see through this analysis that it is a combination of interaction effects and class imbalance that affects both classification and feature importance methods in the biosignature dataset. This has immediate implications for the deployment of ML methods for astrobiology missions: for real complex datasets, as geochemical isotopic data for biosignatures is, the ability to detect statistical interactions is a significant advantage.</p>
<p>For RF permutation importance, the sample must be part of the training data, meaning that a new RF model must be trained to generate local-RF variable importance scores for a single new test sample. Local-RF does better with class imbalance and the small biosignature dataset in terms of separating true and false scores, but because it is a method used during classifier training, the nature of the model could change during the re-training process itself. For the analysis of a single new sample, the change to the base classifier is expected to be negligible; however, if many new samples are introduced in order to generate local-RF importance scores, this could significantly alter the model from the one that originally classified the sample, altering the explainability and in a worst-case scenario, the truth of the outcome variable.</p>
<p>As mentioned in <xref ref-type="sec" rid="s3">Section 3</xref>, the practice of diagnosing false samples using a TLS method requires appropriate thresholds for the RF probability and the TLS to flag samples (see <xref ref-type="table" rid="T2">Table 2</xref>). While the probability and TLS thresholds used in our comparative analysis are based on properties specific to our training data, both local-NPDR and local-RF can successfully diagnose false predictions using various TLS and probability thresholds. Different TLS and probability thresholds result in different numbers of samples being quarantined, some of which will be true predictions that have a low TLS and/or low prediction probability. Future work will incorporate data-driven statistical thresholds such as two-mode Gaussian mixture modeling, where the two modes represent true and false predictions. One way to decide a reasonable TLS threshold is to analyze the distribution of scores in the training datasets for each local importance method and decide a threshold that will minimize samples being flagged in the training data (<xref ref-type="sec" rid="s12">Supplementary Figures S6&#x2013;S8</xref>). Regardless of the thresholding method, the acceptable risk for false predictions will be user and application dependent. For example, in terms of biosignature detection for a future astrobiology mission to an OW, there may be very little tolerance for a false ML prediction. Researchers may use a much stricter RF probability threshold than 75%, and a lower TLS to diagnose potential false predictions. The tradeoff of using a stricter threshold is an increase in false negatives (true predictions quarantined). Additionally, knowledge of statistical interactions and the relative importance of each variable (<italic>e.g.,</italic> as quantified in NPDR&#x2019;s epistasis rank) for each prediction type may be incorporated into an analysis of a ML prediction, ensuring a more robust explanation of the likelihood of a true or false prediction. This type of nuanced analysis can be leveraged to preserve mission resources for future astrobiology missions by increasing the fidelity of the local-NPDR false prediction diagnosis method.</p>
<p>In the current study we show that local-RF variable importance benefits from global-NPDR feature selection. In our application example, local-RF flagged fewer true predictions potentially as false than local-NPDR, enabled by global-NPDR feature selection. It has been previously shown that the RF biosignature classifier benefits from global-NPDR feature selection using an URFP distance metric (<xref ref-type="bibr" rid="B4">Clough et al., 2025</xref>). Since RF has no statistical threshold for limiting the number of variables, if feature selection were not performed, the user would have to decide which of the &#x223c;100 features (in our example datasets) to use in a local importance analysis. Using all &#x223c;100 features, some of which are noise or highly correlated, is likely to result in overfitting, potentially compromising the performance of the local-RF false prediction diagnosis method. The LASSO version of NPDR provides a feature selection threshold that helps local-RF determine local variable importance and therefore to detect false predictions.</p>
<p>Global-NPDR feature selection also allows the RF classifiers to be computationally more lightweight than if the classifier were required to use the full variable space. In a case where training a new model is required&#x2013;as is the case for the local-RF variable importance method&#x2013;that process is much less computationally intensive. This has implications for applications such as onboard learning in automated space exploration. NDPR variable importance methods, both global and local, can contribute to ensuring ML models and data products are an appropriate size for use on flight computers while maintaining accuracy and increasing interpretability.</p>
<p>It is therefore the best practice to use some form of feature selection, like global-NPDR with LASSO penalty. This approach is sensitive to statistical interactions in high-dimensional datasets, making it a natural choice in complex real datasets such as IRMS measurements for astrobiology, or gene expression data for disease prediction. The URFP distance metric adds an additional ability to construct a neighborhood in a non-isotropic variable space, an advantage over traditional distance metrics. NPDR provides additional information about each variable&#x2019;s contributions in terms of main effects and interactions, which enables a more in-depth analysis of each prediction based on the local-NPDR importance scores. This allows us to gain much more than a prediction label for each experimental sample we wish to classify. We can start to articulate how the black box RF is using individual variables to inform classification, and exactly how particular variables may be fooling the model in the case of false predictions. The implications for this increased understanding in the search for OW biosignatures are that we can encode our quantitative understanding of variable effects and each variable&#x2019;s ability to classify samples of a particular class into our science autonomy framework for exploration.</p>
</sec>
<sec sec-type="conclusion" id="s5">
<title>5 Conclusion</title>
<p>One of the primary goals of science is to explain <italic>how</italic> and <italic>why</italic>, however black box ML models in scientific applications are antithetical to this goal. We develop a new ML explanation tool, local-NPDR, and test it on three simulated datasets and one real dataset. Two of the simulated datasets are class imbalanced and one has decreased main effects. The real biosignature-analog dataset has a small sample size, is similarly imbalanced, and is known to have lower variable main effects. Our local-NPDR ML tool can be used to help explain why a single sample is predicted to be in a given class based on the variable importance weights. These weights are computed based on their ability to model the contrastive probability (hits and misses) for samples in the target sample&#x2019;s neighborhood. The sign and magnitude of the NPDR TLS of globally important variables can be used for diagnosing the likelihood of the sample prediction (biotic or abiotic) to be true or false. In conjunction with single sample RF prediction probability, local-NPDR can be a useful tool for future astrobiology missions. The consensus of local-NPDR and local-RF importance methods could improve the ability to detect false samples and discriminate them from flagged true predictions.</p>
<p>Local-NPDR variable importance has the potential advantage of being calculated independently of the classifier. The ability to apply a method that is independent from (agnostic to) the ML classifier to diagnose false predictions can be an advantage as it avoids biases in explainability caused by using the local importance method native to the classifier. If local-NPDR and local-RF were used together, the likelihood of successfully quarantining most if not all falsely predicted samples significantly increases. However, potential limitations of all methods should be considered in high-risk applications. If new samples are very different from samples used to train the models, numeric instability in the local regression may result from distances that are too large between neighboring samples. While this is a potential limitation if not considered, with awareness it can be turned into an indication that the model being used is no longer appropriate, an important conclusion in any high-risk domain.</p>
<p>Being independent of the classifier could also be a potential limitation, since it will not be a perfect explainer for the classifier. In other words, it may be preferrable to have a local explainer that explains the mechanisms of a particular classifier. We limit this by training the RF classifier with global-NPDR selected features, and then local-NPDR will be informative as to whether the outcome label generated by the classifier matches what is expected by the algorithm in the context of a neighborhood of samples in the global-NPDR feature space. Another way to link local-NDPR to a specific classifier could be to use nested cross-validation feature selection (<xref ref-type="bibr" rid="B22">Parvandeh et al., 2020</xref>). Since local-NPDR requires a distance matrix, it still requires access to the training data, like local-RF. However, no model re-training or hyperparameter tuning is needed. Local-NPDR showed evidence of being sensitive to imbalance and small sample sizes while local-RF was less susceptible. Despite these challenges, local-NPDR performs similarly well to local-RF in diagnosing false test predictions in the biosignature dataset and in the simulated datasets (see <xref ref-type="table" rid="T2">Table 2</xref>).</p>
<p>Class imbalance can affect both global and local feature importance methods. Notably, the simulated dataset with class balance and an increased sample size contained zero noise features in the global-NPDR-LURF selected features space and was the best performing simulated dataset for local-NPDR analyses. An important area of future research will be to improve performance for imbalanced data. Additional future work will extend NPDR and local-NPDR to non-tabular data such as images or time series using representation learning, both of which would enable a lightweight version of traditionally computationally intensive methods with added interpretability. Another potential limitation is the need to have a neighborhood of labeled samples, rather than a statistical model, to compute local-NPDR score of a single sample. As mentioned, local-NPDR could be combined with classifier-specific local importance methods, if available, to further improve the detection of false predictions.</p>
<p>In addition to improvements in ML algorithms and explainability, improvements in biosignature detection require close attention to data collection. Both scientific and ML models inherently include some degree of bias, as they rely on initial assumptions or hypotheses before data collection. The resulting data with their biases are then used to identify generalizable patterns. Such assumptions, for example, make it challenging to develop fully agnostic biosignature models from laboratory data. However, both science and ML can guard against biases and discover more general models by making predictions on new data that test the limitations of the existing theory or model. For instance, precise measurements of blackbody radiation exposed shortcomings of classical mechanics (the ultraviolet catastrophe), ultimately leading to a more general atomic theory (quantum mechanics) (<xref ref-type="bibr" rid="B8">Gamow, 1985</xref>). Similarly, in ML, using data outside the training domain can identify limits of a model&#x2019;s validity for prediction. For example, depending on the initial training data, an ocean world (OW) biosignature model may not be valid for certain ranges and combinations of temperature, volatiles, pressure, and salinity.</p>
<p>Both global- and local-NPDR feature importance methods add to the explainability tools currently available for ML methods, which is especially important for high-risk prediction domains. For such ML applications, it is expected that researchers (humans in the loop) will work with the ML algorithm output to ultimately reach the most informed conclusion possible about the prediction in question and mitigate risks associated with false predictions. It is essential to ensure the training data is representative of the deployment environment and its limitations understood, that models are responsibly trained and validated, that models limit noise features and highly correlated features, and tools are included that can help understand individual predictions and diagnose false predictions. Producing a prediction is only a first step in ML; as in science, it is more important to understand the nature of the prediction, how it was made, and whether it should be trusted. Our ML tools take this step in providing explainability.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The data and software supporting the analysis in this study are freely available at <ext-link ext-link-type="uri" xlink:href="https://github.com/insilico/OceanWorldsLocalNPDR">https://github.com/insilico/OceanWorldsLocalNPDR</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>LC: Conceptualization, Formal Analysis, Investigation, Methodology, Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review and editing. JM: Data curation, Writing &#x2013; review and editing. LS: Data curation, Writing &#x2013; review and editing. VD: Writing &#x2013; review and editing. BT: Data curation, Funding acquisition, Writing &#x2013; review and editing. BM: Conceptualization, Funding acquisition, Methodology, Software, Supervision, Writing &#x2013; original draft, Writing &#x2013; review and editing.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. The authors would like to thank Jingyi Chen and the Department of Geosciences at the University of Tulsa for institutional support and Steven Bates in the Stable 1sotope Laboratory at NASA Goddard Space Flight Center (GSFC) for assistance in understanding Isodat<sup>&#xa9;</sup> output. Data collection was funded through a NASA Oklahoma Established Program to Stimulate Competitive Research (EPSCoR) Research Initiation Grant (PI: BT). Data analysis and ML efforts were funded primarily through a NASA GSFC Internal Scientist Funding Model (ISFM) Fundamental Laboratory Research (FLaRe) award &#x201c;Machine Learning of Ocean Worlds Laboratory Analog Seawater Volatiles&#x201d; (PI: BT) and NASA Oklahoma EPSCoR Infrastructure Development, &#x201c;Machine Learning Ocean World Biosignature Detection from Mass Spec&#x201d; (PI: BMcK). NASA Oklahoma Established Program to Stimulate Competitive Research (EPSCoR) Infrastructure Development, &#x201c;Biosignature Detection of Solar System Ocean Worlds using Science-Guided Machine Learning.&#x201d; (PI: BMcK). Grant 80NSSC24M0109.</p>
</sec>
<sec sec-type="COI-statement" id="s9">
<title>Conflict of interest</title>
<p>Author LC was employed by Aurora Engineering. Author VD was employed by Tyto Athene LLC.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s12">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fspas.2025.1651953/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fspas.2025.1651953/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="Supplementaryfile1.docx" id="SM1" mimetype="application/docx" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ali</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Abuhmed</surname>
<given-names>T.</given-names>
</name>
<name>
<surname>El-Sappagh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Muhammad</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Alonso-Moral</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Confalonieri</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2023</year>). <article-title>Explainable artificial intelligence (XAI): what we know and what is left to attain trustworthy artificial intelligence</article-title>. <source>Inf. Fusion</source> <volume>99</volume>, <fpage>101805</fpage>. <pub-id pub-id-type="doi">10.1016/j.inffus.2023.101805</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Breiman</surname>
<given-names>L.</given-names>
</name>
</person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume> (<issue>1</issue>), <fpage>5</fpage>&#x2013;<lpage>32</lpage>. <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choubin</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Jaafari</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Henareh</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Karimi</surname>
<given-names>O.</given-names>
</name>
<name>
<surname>Sajedi Hosseini</surname>
<given-names>F.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Explainable artificial intelligence (XAI) for interpreting predictive models and key variables in flood susceptibility</article-title>. <source>Results Eng.</source> <volume>27</volume>, <fpage>105976</fpage>. <pub-id pub-id-type="doi">10.1016/j.rineng.2025.105976</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clough</surname>
<given-names>L. A.</given-names>
</name>
<name>
<surname>Da Poian</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Major</surname>
<given-names>J. D.</given-names>
</name>
<name>
<surname>Seyler</surname>
<given-names>L. M.</given-names>
</name>
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Theiling</surname>
<given-names>B. P.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Interpretable machine learning biosignature detection from Ocean worlds analogue CO2 isotopologue data</article-title>. <source>Earth Space Sci.</source> <volume>12</volume> (<issue>3</issue>), <fpage>e2024EA003966</fpage>. <pub-id pub-id-type="doi">10.1029/2024EA003966</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Da Poian</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Theiling</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Lyness</surname>
<given-names>E.</given-names>
</name>
<name>
<surname>Burtt</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Azari</surname>
<given-names>A. R.</given-names>
</name>
<name>
<surname>Pasterski</surname>
<given-names>J.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Science autonomy using machine learning for astrobiology</article-title>. <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2504.00709</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Davis</surname>
<given-names>N. A.</given-names>
</name>
<name>
<surname>Crowe</surname>
<given-names>J. E.</given-names>
</name>
<name>
<surname>Pajewski</surname>
<given-names>N. M.</given-names>
</name>
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
</person-group> (<year>2010</year>). <article-title>Surfing a genetic association interaction network to identify modulators of antibody response to smallpox vaccine</article-title>. <source>Genes and Immun.</source> <volume>11</volume> (<issue>8</issue>), <fpage>630</fpage>&#x2013;<lpage>636</lpage>. <pub-id pub-id-type="doi">10.1038/gene.2010.37</pub-id>
<pub-id pub-id-type="pmid">20613780</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dawkins</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Multivariate optimization of k for k-Nearest-Neighbor feature selection with dichotomous outcomes: complex associations, class imbalance, and application to RNA-Seq in major depressive disorder</article-title>. <source>IEEE Trans. Comput. Biol. Bioinforma.</source> <volume>22</volume> (<issue>01</issue>), <fpage>39</fpage>&#x2013;<lpage>51</lpage>. <pub-id pub-id-type="doi">10.1109/TCBBIO.2024.3494599</pub-id>
<pub-id pub-id-type="pmid">40811240</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Gamow</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>1985</year>). <source>Thirty years that shook physics: the story of quantum theory</source>. <publisher-name>Dover Publications, Inc</publisher-name>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lareau</surname>
<given-names>C. A.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Oberg</surname>
<given-names>A. L.</given-names>
</name>
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Differential co-expression network centrality and machine learning feature selection for identifying susceptibility hubs in networks with scale-free structure</article-title>. <source>BioData Min.</source> <volume>8</volume> (<issue>1</issue>), <fpage>5</fpage>. <pub-id pub-id-type="doi">10.1186/s13040-015-0040-x</pub-id>
<pub-id pub-id-type="pmid">25685197</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Le</surname>
<given-names>T. T.</given-names>
</name>
<name>
<surname>Dawkins</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Nearest-neighbor projected-distance regression (NPDR) for detecting network interactions with adjustments for multiple tests and confounding</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>9</issue>), <fpage>2770</fpage>&#x2013;<lpage>2777</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa024</pub-id>
<pub-id pub-id-type="pmid">31930389</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Linardatos</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Papastefanopoulos</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Kotsiantis</surname>
<given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Explainable AI: a review of machine learning interpretability methods</article-title>. <source>Entropy</source> <volume>23</volume> (<issue>1</issue>), <fpage>18</fpage>. <pub-id pub-id-type="doi">10.3390/e23010018</pub-id>
<pub-id pub-id-type="pmid">33375658</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname>
<given-names>S. M.</given-names>
</name>
<name>
<surname>Erion</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>DeGrave</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Prutkin</surname>
<given-names>J. M.</given-names>
</name>
<name>
<surname>Nair</surname>
<given-names>B.</given-names>
</name>
<etal/>
</person-group> (<year>2020</year>). <article-title>From local explanations to global understanding with explainable AI for trees</article-title>. <source>Nat. Mach. Intell.</source> <volume>2</volume> (<issue>1</issue>), <fpage>56</fpage>&#x2013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1038/s42256-019-0138-9</pub-id>
<pub-id pub-id-type="pmid">32607472</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Reif</surname>
<given-names>D. M.</given-names>
</name>
<name>
<surname>Ritchie</surname>
<given-names>M. D.</given-names>
</name>
<name>
<surname>Moore</surname>
<given-names>J. H.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Machine learning for detecting gene-gene interactions</article-title>. <source>Appl. Bioinforma.</source> <volume>5</volume> (<issue>2</issue>), <fpage>77</fpage>&#x2013;<lpage>88</lpage>. <pub-id pub-id-type="doi">10.2165/00822942-200605020-00002</pub-id>
<pub-id pub-id-type="pmid">16722772</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>Jr</surname>
<given-names>J. E. C.</given-names>
</name>
<name>
<surname>Guo</surname>
<given-names>J.</given-names>
</name>
<name>
<surname>Tian</surname>
<given-names>D.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Capturing the spectrum of interaction effects in genetic association studies by simulated evaporative cooling network analysis</article-title>. <source>PLOS Genet.</source> <volume>5</volume> (<issue>3</issue>), <fpage>e1000432</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pgen.1000432</pub-id>
<pub-id pub-id-type="pmid">19300503</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
<name>
<surname>White</surname>
<given-names>B. C.</given-names>
</name>
<name>
<surname>Grill</surname>
<given-names>D. E.</given-names>
</name>
<name>
<surname>Li</surname>
<given-names>P. W.</given-names>
</name>
<name>
<surname>Kennedy</surname>
<given-names>R. B.</given-names>
</name>
<name>
<surname>Poland</surname>
<given-names>G. A.</given-names>
</name>
<etal/>
</person-group> (<year>2013</year>). <article-title>ReliefSeq: a gene-wise Adaptive-K nearest-neighbor feature selection tool for finding gene-gene interactions and main effects in mRNA-Seq gene expression data</article-title>. <source>PLOS ONE</source> <volume>8</volume> (<issue>12</issue>), <fpage>e81527</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0081527</pub-id>
<pub-id pub-id-type="pmid">24339943</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Montavon</surname>
<given-names>G.</given-names>
</name>
<name>
<surname>Samek</surname>
<given-names>W.</given-names>
</name>
<name>
<surname>M&#xfc;ller</surname>
<given-names>K.-R.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Methods for interpreting and understanding deep neural networks</article-title>. <source>Digit. Signal Process.</source> <volume>73</volume>, <fpage>1</fpage>&#x2013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1016/j.dsp.2017.10.011</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Nagel</surname>
<given-names>E.</given-names>
</name>
</person-group> (<year>1979</year>). <source>The structure of science: problems in the logic of scientific explanation</source>. <publisher-loc>Indianapolis, Indiana</publisher-loc>: <publisher-name>Hackett Publishing Co</publisher-name>. <comment>Available online at: <ext-link ext-link-type="uri" xlink:href="https://hackettpublishing.com/philosophy/philosophy-science/the-structure-of-science">https://hackettpublishing.com/philosophy/philosophy-science/the-structure-of-science</ext-link>.</comment>
</citation>
</ref>
<ref id="B18">
<citation citation-type="book">
<collab>National Academies of Sciences, Engineering, and Medicine</collab> (<year>2019</year>). &#x201c;<article-title>Biosignature identification and interpretation</article-title>,&#x201d; in <source>An astrobiology strategy for the search for life in the universe</source> (<publisher-name>National Academies Press US</publisher-name>). <pub-id pub-id-type="doi">10.17226/25252</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Navarra</surname>
<given-names>G. G.</given-names>
</name>
<name>
<surname>Deutsch</surname>
<given-names>C.</given-names>
</name>
<name>
<surname>Mamalakis</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>Margolskee</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>MacGilchrist</surname>
<given-names>G.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Long term predictability of Southern Ocean surface nutrients using explainable neural networks</article-title>. <source>J. Geophys. Res. Mach. Learn. Comput.</source> <volume>2</volume> (<issue>2</issue>), <fpage>e2024JH000268</fpage>. <pub-id pub-id-type="doi">10.1029/2024JH000268</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Noordijk</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Garcia Gomez</surname>
<given-names>M. L.</given-names>
</name>
<name>
<surname>ten Tusscher</surname>
<given-names>K. H. W. J.</given-names>
</name>
<name>
<surname>de Ridder</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>van Dijk</surname>
<given-names>A. D. J.</given-names>
</name>
<name>
<surname>Smith</surname>
<given-names>R. W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>The rise of scientific machine learning: a perspective on combining mechanistic modelling with machine learning for systems biology</article-title>. <source>Front. Syst. Biol.</source> <volume>4</volume>, <fpage>1407994</fpage>. <pub-id pub-id-type="doi">10.3389/fsysb.2024.1407994</pub-id>
<pub-id pub-id-type="pmid">40809147</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Novielli</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Magarelli</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Romano</surname>
<given-names>D.</given-names>
</name>
<name>
<surname>Di Bitonto</surname>
<given-names>P.</given-names>
</name>
<name>
<surname>Stellacci</surname>
<given-names>A. M.</given-names>
</name>
<name>
<surname>Monaco</surname>
<given-names>A.</given-names>
</name>
<etal/>
</person-group> (<year>2025</year>). <article-title>Leveraging explainable AI to predict soil respiration sensitivity and its drivers for climate change mitigation</article-title>. <source>Sci. Rep.</source> <volume>15</volume> (<issue>1</issue>), <fpage>12527</fpage>. <pub-id pub-id-type="doi">10.1038/s41598-025-96216-y</pub-id>
<pub-id pub-id-type="pmid">40216855</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Parvandeh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Yeh</surname>
<given-names>H.-W.</given-names>
</name>
<name>
<surname>Paulus</surname>
<given-names>M. P.</given-names>
</name>
<name>
<surname>McKinney</surname>
<given-names>B. A.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Consensus features nested cross-validation</article-title>. <source>Bioinformatics</source> <volume>36</volume> (<issue>10</issue>), <fpage>3093</fpage>&#x2013;<lpage>3098</lpage>. <pub-id pub-id-type="doi">10.1093/bioinformatics/btaa046</pub-id>
<pub-id pub-id-type="pmid">31985777</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ribeiro</surname>
<given-names>M. T.</given-names>
</name>
<name>
<surname>Singh</surname>
<given-names>S.</given-names>
</name>
<name>
<surname>Guestrin</surname>
<given-names>C.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Why should I trust you? explaining the predictions of any classifier</article-title>. <pub-id pub-id-type="doi">10.48550/arXiv.1602.04938</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roscher</surname>
<given-names>R.</given-names>
</name>
<name>
<surname>Bohn</surname>
<given-names>B.</given-names>
</name>
<name>
<surname>Duarte</surname>
<given-names>M. F.</given-names>
</name>
<name>
<surname>Garcke</surname>
<given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Explainable machine learning for scientific insights and discoveries</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>42200</fpage>&#x2013;<lpage>42216</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2976199</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Theiling</surname>
<given-names>B. P.</given-names>
</name>
<name>
<surname>Chou</surname>
<given-names>L.</given-names>
</name>
<name>
<surname>Da Poian</surname>
<given-names>V.</given-names>
</name>
<name>
<surname>Battler</surname>
<given-names>M.</given-names>
</name>
<name>
<surname>Raimalwala</surname>
<given-names>K.</given-names>
</name>
<name>
<surname>Arevalo</surname>
<given-names>R.</given-names>
</name>
<etal/>
</person-group> (<year>2022</year>). <article-title>Science autonomy for ocean worlds astrobiology: a perspective</article-title>. <source>Astrobiology</source> <volume>22</volume> (<issue>8</issue>), <fpage>901</fpage>&#x2013;<lpage>913</lpage>. <pub-id pub-id-type="doi">10.1089/ast.2021.0062</pub-id>
<pub-id pub-id-type="pmid">35507950</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tibshirani</surname>
<given-names>R.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Regression shrinkage and selection <italic>via</italic> the lasso</article-title>. <source>J. R. Stat. Soc. Ser. B Methodol.</source> <volume>58</volume> (<issue>1</issue>), <fpage>267</fpage>&#x2013;<lpage>288</lpage>. <pub-id pub-id-type="doi">10.1111/j.2517-6161.1996.tb02080.x</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wright</surname>
<given-names>M. N.</given-names>
</name>
<name>
<surname>Ziegler</surname>
<given-names>A.</given-names>
</name>
<name>
<surname>K&#xf6;nig</surname>
<given-names>I. R.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Do little interactions get lost in dark random forests</article-title>. <source>BMC Bioinforma.</source> <volume>17</volume> (<issue>1</issue>), <fpage>145</fpage>. <pub-id pub-id-type="doi">10.1186/s12859-016-0995-8</pub-id>
<pub-id pub-id-type="pmid">27029549</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zou</surname>
<given-names>H.</given-names>
</name>
<name>
<surname>Hastie</surname>
<given-names>T.</given-names>
</name>
</person-group> (<year>2005</year>). <article-title>Regularization and variable selection <italic>via</italic> the elastic net</article-title>. <source>J. R. Stat. Soc. Ser. B Stat. Methodol.</source> <volume>67</volume> (<issue>2</issue>), <fpage>301</fpage>&#x2013;<lpage>320</lpage>. <pub-id pub-id-type="doi">10.1111/j.1467-9868.2005.00503.x</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>