<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article article-type="research-article" dtd-version="2.3" xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Phys.</journal-id>
<journal-title>Frontiers in Physics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Phys.</abbrev-journal-title>
<issn pub-type="epub">2296-424X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">783765</article-id>
<article-id pub-id-type="doi">10.3389/fphy.2022.783765</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Physics</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A Unified Approach to Analysis of MRI Radiomics of Glioma Using Minimum Spanning Trees</article-title>
<alt-title alt-title-type="left-running-head">Simon et al.</alt-title>
<alt-title alt-title-type="right-running-head">Minimum Spanning Trees For Radiomics</alt-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Simon</surname>
<given-names>Olivier B.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Jain</surname>
<given-names>Rajan</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Choi</surname>
<given-names>Yoon-Seong</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>G&#xf6;rg</surname>
<given-names>Carsten</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1151165/overview"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Suresh</surname>
<given-names>Krithika</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Severn</surname>
<given-names>Cameron</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1196346/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Ghosh</surname>
<given-names>Debashis</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="corresp" rid="c001">&#x2a;</xref>
<uri xlink:href="https://loop.frontiersin.org/people/809667/overview"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Biostatistics and Informatics</institution>, <institution>Colorado School of Public Health</institution>, <institution>University of Colorado Anschutz Medical Campus</institution>, <addr-line>Aurora</addr-line>, <addr-line>CO</addr-line>, <country>United States</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Department of Radiology</institution>, <institution>NYU Grossman School of Medicine</institution>, <addr-line>New York</addr-line>, <addr-line>NY</addr-line>, <country>United States</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Neurosurgery</institution>, <institution>NYU Grossman School of Medicine</institution>, <addr-line>New York</addr-line>, <addr-line>NY</addr-line>, <country>United States</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Radiological Sciences Academic Clinical Programme</institution>, <institution>Duke-NUS Medical School</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Radiology</institution>, <institution>Yonsei University College of Medicine</institution>, <addr-line>Seoul</addr-line>, <country>South Korea</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Pediatrics</institution>, <institution>Section of Endocrinology</institution>, <institution>School of Medicine</institution>, <institution>University of Colorado Anschutz Medical Campus</institution>, <addr-line>Aurora</addr-line>, <addr-line>CO</addr-line>, <country>United States</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>
<bold>Edited by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1063589/overview">Daniel Rodriguez Gutierrez</ext-link>, Nottingham University Hospitals NHS Trust, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>
<bold>Reviewed by:</bold> <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/654727/overview">Zhiwei Ji</ext-link>, Nanjing Agricultural University, China</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1445165/overview">Roberto Gatta</ext-link>, University of Brescia, Italy</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/62746/overview">Enrico Capobianco</ext-link>, University of Miami, United States</p>
</fn>
<corresp id="c001">&#x2a;Correspondence: Debashis Ghosh, <email>debashis.ghosh@cuanschutz.edu</email>
</corresp>
<fn fn-type="other">
<p>This article was submitted to Medical Physics and Imaging, a section of the journal Frontiers in Physics</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>05</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>10</volume>
<elocation-id>783765</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>09</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>12</day>
<month>04</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2022 Simon, Jain, Choi, G&#xf6;rg, Suresh, Severn and Ghosh.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Simon, Jain, Choi, G&#xf6;rg, Suresh, Severn and Ghosh</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Radiomics has shown great promise in detecting important genetic markers involved in cancers such as gliomas, as specific mutations produce subtle but characteristic changes in tumor texture and morphology. In particular, mutations in IDH (isocitrate dehydrogenase) are well-known to be important prognostic markers in glioma patients. Most classification approaches using radiomics, however, involve complex hand-crafted feature sets or &#x201c;black-box&#x201d; methods such as deep neural networks, and therefore lack interpretability. Here, we explore the application of simple graph-theoretical methods based on the minimum-spanning tree (MST) to radiomics data, in order to detect IDH mutations in gliomas. This is done using a hypothesis testing approach. The methods are applied to an fMRI dataset on <italic>n</italic> &#x3d; 413 patients. We quantify the significance of the group-wise difference between mutant and wild-type using the MST edge-count testing methodology of Friedman and Rafsky. We apply network theory-based centrality measures on MSTs to identify the most representative patients. We also propose a simple and rapid dimensionality-reduction method based on k-MSTs. Combined with the centrality measures, the latter method produces readily interpretable 2D maps that reveal distinct IDH, non-IDH, and IDH-like groupings.</p>
</abstract>
<kwd-group>
<kwd>medical imaging</kwd>
<kwd>biostatistics</kwd>
<kwd>genotype-phenotype correlation</kwd>
<kwd>tree-based methodology</kwd>
<kwd>data visualization</kwd>
</kwd-group>
<contract-sponsor id="cn001">National Cancer Institute<named-content content-type="fundref-id">10.13039/100000054</named-content>
</contract-sponsor>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>The advent of widespread medical imaging, large imaging datasets, and large-scale inexpensive computing power has ushered in an era of unprecedented resources for medical image analysis [<xref ref-type="bibr" rid="B1">1</xref>]. Cancers can now be automatically detected and staged from histopathology images, or from clinical imaging datasets such as MRI, CT or PET data. In particular, considerable success has been achieved using complex computer-derived image-analysis features derived from such data as input for advanced statistical and machinelearning methods. This approach, known as &#x201c;radiomics&#x201d;, offers the potential to take into account multiple features of the image not detected by human observers and hence also avoiding the issue of inter-observer variability [<xref ref-type="bibr" rid="B2">2</xref>].</p>
<p>Genotyping of gliomas is difficult and invasive, as it requires biopsy of brain tissue. While some genetic correlates of cancer prognosis, such as MGMT promoter methylation, have not shown strong correlation with radiomics features [<xref ref-type="bibr" rid="B3">3</xref>], other mutation types are associated with marked differences in radiomic profiles&#x2014;although considerable variability between studies exists. In particular, isocitrate dehydrogenase (IDH) mutations are found in 5&#x2013;13% of glioblastomas and are strongly correlated with radiomics features [<xref ref-type="bibr" rid="B3">3</xref>].</p>
<p>Current automated methods for visual or radiomic genotyping of gliomas increasingly depend on deep neural network methods and pipelines, often using off-the-shelf architectures such as ResNet for detection and then classification [<xref ref-type="bibr" rid="B4">4</xref>]. Still other studies have made use of random forest methods for genotyping, in combination with CNN-based methods for tumor segmentation [<xref ref-type="bibr" rid="B3">3</xref>]. Alongside neural networks, more traditional &#x201c;hand-crafted&#x201d; features, involving human-defined combinations of pixel-level image analysis methods such as gray-level co-occurrence matrices (GLCMs), represent a second still-vibrant branch of radiomics analysis [<xref ref-type="bibr" rid="B5">5</xref>]. Handcrafted features often have the benefit of imparting greater interpretability to radiomics analyses; on the other hand, since neural network models are considered by some to be more free from human bias, current state-of-the-art radiomics methods frequently combine both [<xref ref-type="bibr" rid="B6">6</xref>].The area under the curve (AUC) is a typical metric for evaluation for these approaches, with values around 0.85&#x2212;0.95 representing the state-of-the-art as of this writing [<xref ref-type="bibr" rid="B2">2</xref>]. Other measures such as F1 score, sensitivity and specificity are also common. However, while useful for gauging performance, these measures do little to provide an intuitive understanding of the structure of the underlying data, or the reasons for the classifier outputs&#x2014;a problem which is particularly serious with neural networks, which with their many millions of automatically learned parameters are often considered to be &#x201c;black-boxes&#x201d; [<xref ref-type="bibr" rid="B7">7</xref>]. While deep learning is the current state-of-the art classification technique, we do note that other modeling procedures could be used, such as logistic regression, support vector machines or L1-penalized regression approaches, among many others.</p>
<p>Minimum spanning trees (MSTs) are graph-theoretic structures in which a set of data-points or &#x201c;nodes&#x201d; are connected into a single component using the minimum possible total connection distance [<xref ref-type="bibr" rid="B8">8</xref>]. Notably, MSTs, while easy to compute, are capable of representing key statistical properties of highly complex datasets in a vastly simplified format that is also readily amenable to lower-dimensional (even 2D) visualization. This renders them applicable to understanding a range of systems, such as gene expression, transportation networks and brain connectivity [<xref ref-type="bibr" rid="B9">9</xref>, <xref ref-type="bibr" rid="B10">10</xref>]. Furthermore, node centrality measures--which aim at measuring the &#x201c;importance&#x201d; of a given node to the overall network structure--are readily applicable to the MST [<xref ref-type="bibr" rid="B11">11</xref>]. Therefore, MST and other graph-based approaches may offer an appealing and complementary alternative to the &#x2018;blpredictions given by neural-networks.</p>
<p>In addition to being easy to calculate, the MST of a high-dimensional dataset also comes with an attendant hypothesis testing procedure that allows one to assess the significance of the difference between classes. This is the Friedman-Rafsky multivariate runs test (here abbreviated &#x201c;MVR&#x201d;) [<xref ref-type="bibr" rid="B12">12</xref>&#x2013;<xref ref-type="bibr" rid="B14">14</xref>]. Briefly, MVR involves constructing an MST over the pooled data from two different classes, removing the edges that connect different classes, and counting the number of connected components that result. Smaller numbers of connected components indicate greater significance between the classes; this significance, furthermore, can be calculated using a standard normal approximation. Note that our goal here of inference is substantially different from much of the radiomics literature described above, which is focused on classification performance.</p>
<p>The k-MST is a simple extension of the MST, found by repeating the MST algorithm k times, each time excluding any connections chosen in prior iterations [<xref ref-type="bibr" rid="B14">14</xref>]. This allows a richer level of connectivity information which in turn can improve statistical test results such as edge-counting [<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B14">14</xref>]. At the same time, like an MST, a k-MST is a uniquely defined mathematical structure that can be calculated from any given point-set without requiring any user-tuned parameters; thus, use of the k-MST may greatly ameliorate one of the common concerns regarding &#x201c;handcrafted&#x201d; radiomics features, namely that of lower reproducibility stemming from bias in the feature design [<xref ref-type="bibr" rid="B6">6</xref>].</p>
<p>In the present work, we use the k-MST as a representation of the underlying structure of multivariate radiomics data, randomly embed it in a 2D region, and apply a simple 2D force-directed layouts methodology whereby nodes that are directly connected in the k-MST experience an attractive force. To avoid the expensive process of calculating repulsive forces between non-connected nodes, an isotropic expansion or &#x201c;reflation&#x201d; is carried out after each iteration, to counteract the tendency of a wholly-attractive configuration of forces to collapse to a point.</p>
<p>Because the k-MST contains only a small fraction of the possible pairwise connections between nodes, and because there are no explicit repulsive forces to calculate, our method allows rapid creation of 2D representations of arbitrarily high-dimensional radiomics datasets. Importantly, we find this method consistently converges to configurations which effectively segregate the IDH and non-IDH patients&#x2014;especially when combined with the results of node centrality measures. This suggests possible wide applications of MST-based methods in creating &#x201c;explainable&#x201d; maps of radiomic data with respect to tumor genotype.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>Methods</title>
<p>Our analytic workflows are described in <xref ref-type="fig" rid="F1">Figures 1</xref>,<xref ref-type="fig" rid="F2">2</xref>. Our dataset derives from MRI scans conducted on 413 glioma patients, genotyped as either IDH-mutated (<italic>n</italic> &#x3d; 144) or IDH wild type (<italic>n</italic> &#x3d; 269). The data come from a previously published study [<xref ref-type="bibr" rid="B4">4</xref>]. T2-weighted and fluid-attenuated inversion recovery (FLAIR) MR images of diffuse gliomas (WHO grades II, III and IV) were obtained in DICOM format from After conversion to NIfTI format, T2-weighted images were re-sampled to 1&#xa0;mm isovoxel resolution using the &#x2018;trilinear&#x2019; option from the FLIRT function, while FLAIR images were registered to the T2 images after skull stripping, all using the FMRIB software library (<ext-link ext-link-type="uri" xlink:href="http://fsl.fmrib.ox.ac.uk/fsl/fslwiki/FSL">http://fsl.fmrib.ox.ac.uk/fsl/fslwiki/FSL</ext-link>). Next, image signal intensity was normalized using the WhiteStripe R package. Tumor areas (defined by hyper-intensity in T2 images and edema on FLAIR images) were segmented with semi-automatic methods such as region growing, signal intensity thresholding, and edge detection, with an open-source software (Medical Image Processing, Analysis and Visualization, <ext-link ext-link-type="uri" xlink:href="https://mipav.cit.nih.gov/">https://mipav.cit.nih.gov/</ext-link>). Segmentations were manually corrected by a neuroradiologist as deemed necessary.</p>
<fig id="F1" position="float">
<label>FIGURE 1</label>
<caption>
<p>Flowchart for analysis using centrality maps.</p>
</caption>
<graphic xlink:href="fphy-10-783765-g001.tif"/>
</fig>
<fig id="F2" position="float">
<label>FIGURE 2</label>
<caption>
<p>Flowchart for analysis using k-MST force-directed maps.</p>
</caption>
<graphic xlink:href="fphy-10-783765-g002.tif"/>
</fig>
<p>Once MRI post-processing was completed, 467 radiomics features were calculated per patient using the PyRadiomics suite [<xref ref-type="bibr" rid="B15">15</xref>]. A full list of features used is included under (<xref ref-type="sec" rid="s10">Supplementary Table S4</xref>). All data was centered to zero and normalized by dividing each column by its standard deviation. To account for the possibility of redundancy or overlap among the radiomics features, our MATLAB pipeline provides the option to perform PCA, retaining only those components which together comprise &#x3e;98% of the total variance. This step reduces the number of components from the original 467 to 48. Example results of our pipeline using this PCA step are provided in Supplementary Figures; however, as this step did not dramatically change the character of the results, it was not used in the main study.</p>
<p>Next, using the features as dimensions and each patient as a node, we constructed MSTs over the pooled patient data from both groups and carried out the multivariate runs (MVR) test outlined by Friedman and colleagues [<xref ref-type="bibr" rid="B12">12</xref>](13) [<xref ref-type="bibr" rid="B14">14</xref>]. The Euclidean distance based on the standardized radiomics feature vectors was used to calculate distances between all pairs of subjects. This yields a graph with edge weights based on the distance which is used to construct an MST. For the MVR test, edges connecting dissimilar node-types (i.e., nodes connected from two different groups) are removed, yielding a number of disjoint trees, R. Given two MSTs with N<sub>a</sub> and N<sub>b</sub> nodes, (and N &#x3d; N<sub>a</sub> &#x2b; N<sub>b</sub>), Friedman and Rafsky demonstrate that R is normally distributed, with mean equal to<disp-formula id="equ1">
<mml:math id="m1">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>and variance (conditioned on C, the number of pairs of edges that share a common node in the given MST), equal to<disp-formula id="equ2">
<mml:math id="m2">
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mtext>&#x7c;</mml:mtext>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mo>&#x3d;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>N</mml:mi>
</mml:mrow>
<mml:mi>N</mml:mi>
</mml:mfrac>
<mml:mo>&#x2b;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>N</mml:mi>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>N</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>b</mml:mi>
</mml:msub>
<mml:mo>&#x2b;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:math>
</disp-formula>
</p>
<p>This allows rapid, exact, and direct assessment of the degree of significant similarity between the IDH and non-IDH groups.</p>
<p>Next, a variety of node centrality measures were calculated for each node of the MSTs drawn over the IDH and non-IDH groups separately. Six measures of centrality were assessed for each node included: 1) degree centrality; 2) total degree count of neighbors; 3) through-space closeness; 4) through-tree closeness; 5) betweenness; and 6) eigenvector. Degree centralities are simply the number of other nodes to which the node of interest is directly attached; closeness is the inverse of average distance to all other nodes, either through space or through the MST connections; betweenness indicates the proportion of all the shortest paths between nodes in the MST that pass through the node of interest; and eigenvalue centrality, roughly speaking, combines the concepts of degree and betweenness by relating each MST node to the entries of the principal eigenvector of the MST connectivity matrix [<xref ref-type="bibr" rid="B16">16</xref>].</p>
<p>The k-MST is an extension of the MST, found by repeating the MST algorithm k times, each time excluding all the connections previously chosen. This allows a richer level of connectivity information which in turn can improve statistical test results such as edge-counting. Here, we use the k-MST as a representation of the underlying structure of multivariate data and apply a simple 2D force-directed layout methodology whereby nodes that are directly connected in the k-MST experience an attractive force. &#x201c;Reflation&#x201d; is carried out after each iteration, to counteract the tendency of a wholly-attractive configuration of forces to collapse to a point. Because the k-MST contains only a small fraction of the possible pairwise connections between nodes, and because there are no explicit repulsive forces to calculate, our method allows rapid creation of 2D representations of arbitrarily high-dimensional radiomics datasets.</p>
<p>For the kMST force-directed layouts method for dimension reduction, the steps are as follows: using the distance matrix over the pooled patient nodes, the minimal spanning tree algorithm is iteratively applied, each time setting the distance matrix entries corresponding to chosen edges to a high value so that they are not chosen again. The edges chosen by each successive MST calculation are then saved. For the present work, we used a 5-MST, or 5 iterations.</p>
<p>The method is then initialized by assigning the nodes to random positions within the unit square (we used a uniform distribution was used for this purpose). Next, position updates are iteratively calculated, by summing the attractive &#x2018;forces&#x2019; exerted on each node by only its direct neighbors within the MST. The attractive force is &#x201c;spring-like&#x201d; in that it increases linearly with distance between nodes.</p>
<p>To avoid having to calculate numerous repulsion effects between all nodes not connected in the kMST, we instead implemented an &#x2018;inflationary&#x2019; step: at the end of each iteration the coordinates of the nodes are automatically rescaled to fit just inside the unit square. This inflationary step preserves the configuration changes of each position update while preventing the whole configuration from collapsing to a point.</p>
<p>Two parameters are used to generate the position updates: dEq, the equilibrium distance where attraction between kMST neighbors becomes repulsion with decreasing distance; and kAtt, the relative strength of the attractive force. For this study, we used values of dEq &#x3d; 0.025 and kAtt &#x3d; 0.015.</p>
<p>All calculations (after MRI acquisition and processing, and radiomics feature extraction) were implemented directly using a custom MATLAB pipeline. Scripts used are available on Github at <ext-link ext-link-type="uri" xlink:href="https://github.com/Ghoshlab/OSimonScripts">https://github.com/Ghoshlab/OSimonScripts</ext-link>.</p>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<p>Node centralities for IDH-mutated and IDH-wildtype patients are displayed using six different centrality definitions in <xref ref-type="fig" rid="F3">Figure 3</xref> and <xref ref-type="sec" rid="s10">Supplementary Figure S1</xref>, using node size and color to represent centrality. Notably, in this case, the same small number of nodes were consistently chosen as &#x201c;most central&#x201d;, despite the wide differences in the centrality definitions applied. Specifically, for the IDH-mutated gliomas, Patient 24 was the &#x2018;most central&#x2019; for node degree, node neighbor degree, through-space closeness, and eigenvector centrality, while Patient 80 was the most central in the case of through-tree closeness and betweenness. Among the IDH wild type gliomas, three nodes were prominent: Patient 35 (degree centrality), Patient 37 (neighbor degree, through-tree closeness), and Patient 39 (through-space closeness, eigenvector).</p>
<fig id="F3" position="float">
<label>FIGURE 3</label>
<caption>
<p>MSTs for the IDH mutation positive (<xref ref-type="fig" rid="F3">Figure 3A</xref>) and IDH mutation negative tumors (<xref ref-type="fig" rid="F3">Figure 3B</xref>), with node sizes proportional to degree of the nodes. The <italic>X</italic>- and <italic>Y</italic>-axes represent spatial coordinates for visualization of the minimum spanning trees.</p>
</caption>
<graphic xlink:href="fphy-10-783765-g003.tif"/>
</fig>
<p>These nodes are usually located towards the &#x201c;center&#x201d; of the MST, usually at a junction between several sub-trees. Conversely, the lowest-centrality nodes are invariably found at the edges of the MST, among nodes with only one connection (&#x201c;leaves&#x201d;). These observations confirm that these measures do indeed reflect the intuitive idea of centrality.</p>
<p>Additionally, when PCA reduction was used to decrease the number of features, nearly all the central nodes remained the same, with the sole exception that in the IDH-wildtype gliomas the highest eigenvector centrality shifted from Patient 39 to Patient 35 (<xref ref-type="sec" rid="s10">Supplementary Figure S2</xref>). This close correspondence suggests that the centrality measures and MST algorithms used are robust to complex manipulations and changes of coordinates, such as those which occur using PCA.</p>
<p>The MVR results for our radiomics dataset (<xref ref-type="fig" rid="F4">Figure 4</xref>) showed a clear distinction between the IDH-mutated and IDH-wildtype groups, consistent with prior literature reporting the strong effect of this mutation on radiomics profiles. Beginning with the pooled MST, the number of separate trees that would be expected in the null case (188.6) far exceeds the actual number resulting from the cut (91). This amounts to a difference of &#x2212;9.82 standard deviations, effectively excluding the possibility that the groups differ according to chance. Thus, the MVR test of Friedman and Rafsky rejects the null hypothesis of no difference between the two groups with a <italic>p</italic>-value less than 1 &#xd7; 10<sup>&#x2212;32</sup>.</p>
<fig id="F4" position="float">
<label>FIGURE 4</label>
<caption>
<p>
<bold>(A)</bold> The pooled MST before the hybrid-edge cut; <bold>(B)</bold> A 2-dimensional representation of the subtrees of the MSTs after removing the hybrid edges. The X- and Y-axes represent spatial coordinates for visualization of the minimum spanning trees.</p>
</caption>
<graphic xlink:href="fphy-10-783765-g004.tif"/>
</fig>
<p>As was the case with the node centralities, the MVR test carried out with PCA reduction to 48 features (<xref ref-type="sec" rid="s10">Supplementary Figure S3</xref>) yielded results similar to the original dataset. While the null expectation value of the number of trees remains the same by definition, number of trees from the actual cut (97), and the total standard deviations from the expectation value (&#x2212;9.21) reveal a change of only half a standard deviation despite the PCA manipulation. This again helps establish the robustness of these methods to feature selection.</p>
<p>Applying the kMST-force directed algorithm to the 5-MST drawn over the pooled data-points, we found that the method rapidly and effectively produces readily interpretable visual layouts of the group structure. A representative result is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. The results of three randomly-initialized runs of the algorithm using 3-MST, 5-MST, and 7-MST respectively (<xref ref-type="sec" rid="s10">Supplementary Figures 4A&#x2013;I</xref>) show that, despite the random initialization and the large number of datapoints involved, the final configurations produced by the algorithm are remarkably consistent overall (notwithstanding mirror-symmetry and rotations), and also reveal a clear though not perfect spatial separation between the two genotype groups. As might be expected, the runs using the lowest-complexity k-MST (the 3-MST) show somewhat more variability in the final structure and also a somewhat different final structure from the others, whereas the 5-MST and 7-MST show quite good consistency both between random initializations and between each other. This strongly indicates that the k-MST, despite its much-simplified structure with respect to the full graph, contains the information necessary for meaningful and reproducible dimension reduction of the data and that its local minimum under force calculation is likely unique. Furthermore, as in the previous approaches, the layouts were consistent and quite similar even when PCA reduction was first carried out (data not shown).</p>
<fig id="F5" position="float">
<label>FIGURE 5</label>
<caption>
<p>Representative example of 5-MST force-directed result using PCA-reduced data. IDH-mutated patients are shown in orange, while IDH-wildtype is blue. The algorithm was run for 1800 iterations, with the values of kEq and kAtt held constant. The <italic>X</italic>- and <italic>Y</italic>-axes represent spatial coordinates for visualization of the minimum spanning trees.</p>
</caption>
<graphic xlink:href="fphy-10-783765-g005.tif"/>
</fig>
<p>Given that the k-MST force-directed algorithm did not perfectly separate the two classes&#x2014;more non-IDH nodes are present in the region dominated by IDH than vice-versa--we wondered whether the centrality measures would reasonably reflect the location of the nodes with respect to the overall distribution of class examples within the 2D layout, i.e., whether the nodes closest to the center of their class distribution in the layout would have the highest centralities as well. We found this to be generally the case, though the choice of centrality measure does have an effect. Interestingly, we find that eigenvector centrality gives much lower precedence to the &#x201c;IDH-like&#x201d; non-IDH cases found in the predominantly region, yielding better separation between IDH and wild-type regions when <italic>both</italic> centrality and kMST layout are used (<xref ref-type="sec" rid="s10">Supplementary Figure 4F</xref>). Other measures, particularly betweenness, seem to be much less effective at distinguishing &#x2018;IDH-mimics&#x2019; from the other wild-types&#x2014;there are a few IDH-like cases with relatively high betweenness, but this centrality also yield an even clearer overall divide between the main IDH and wild-type than does eigenvector (<xref ref-type="sec" rid="s10">Supplementary Figure 4E</xref>). Degree-based or closeness-based centrality measures, on the other hand, do not appear to be especially effective at discriminating the central regions of the two classes in the kMST layout, even when the difference is exaggerated by squaring the centrality (<xref ref-type="sec" rid="s10">Supplementary Figures 4A&#x2013;D</xref>).</p>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>In the foregoing, we have demonstrated the feasibility of a simple graph-theoretical toolkit to address the problems presented by large, high-dimensional radiomic datasets. For the example dataset drawn from IDH-mutated and IDH-wildtype glioma patients, we were able to use MST-based methods to establish highly significant differences between the two groups, to identify patients that are most &#x201c;representative&#x201d; of each group using a combination of centrality measures, and to use a simple kMST-based force-directed method to illustrate those centrality measures within the context of a two-dimensional map of the data. Importantly, we find this method converges to a very consistent configuration which can effectively segregate IDH-mutated and IDH wild-type gliomas, especially when combined with centrality measures.</p>
<p>Although there is overlap between the two classes, there is a very clear difference in the overall localization between the two groups. Particularly focusing on the 5-MST and 7-MST&#x2014;which converged consistently to a roughly triangular 2D point distribution&#x2014;we see that IDH-wildtype patients tend to group strongly in one corner of the triangle with almost no IDH-mutated patients present, while at another corner and towards the center of the triangle IDH-mutated cases predominate. Notably, a significant minority of IDH-wildtype patients exhibit IDH-like localization, suggesting that the IDH mutation is sufficient but not necessary for an &#x201c;IDH-like&#x201d; phenotype. This means that stratification by radiomic features alone may be vulnerable to false positives, in the sense that patients with typically IDH-like radiomic features (at least according to our mode of analysis) may nonetheless lack the IDH mutation. Conversely, there might be further subtypes of IDH-wildtype populations to further characterize, although our study is not sufficiently powered for this type of discovery.</p>
<p>We hypothesize this may be due to other mutations or combinations of mutations that partially phenocopy the IDH mutation. Future genetic studies may help elucidate these IDH-mimicking gene combinations, perhaps by looking for epistatic effects on the IDH pathway [<xref ref-type="bibr" rid="B17">17</xref>]. Furthermore, it would be highly worthwhile to track the outcomes of IDH-like patients, to determine whether they in fact share the prognosis generally associated with IDH mutation proper. As noted, our results suggest that the combination of eigenvector centrality and k-MST layouts may be especially useful in distinguishing between wild-type glioma patients that are IDH-like and those that are more &#x2018;typical&#x2019;.</p>
<p>Among the force-directed layouts methods, the Barnes-Hut algorithm, which coalesces sufficiently distant points into a single center of mass by constructing a quadtree based on a distance criterion, may be the best-known means of simplifying force calculations for very large numbers of points [<xref ref-type="bibr" rid="B18">18</xref>]. In our case, however, the use of the k-MST greatly reduces the number of interactions that need to be calculated at each iteration and inherently restrains the calculation only to &#x201c;sufficiently close&#x201d; node pairs, so that the Barnes-Hut approach is unnecessary. The replacement of explicit repulsion term calculations with a simple &#x201c;inflationary&#x201d; step after each iteration also simplifies the overall calculation while likely reducing the chance of the configuration becoming trapped in &#x201c;geometrically-frustrated&#x201d; local minima. Notably, for 5-MST and higher, we saw no evidence of alternative minima for our simulation&#x2014;in all conditions tested, the overall arrangement of points did not differ qualitatively from the overall pattern seen here. It will be interesting to see if this general pattern is observed for different datasets.</p>
<p>The choice of k for the k-MST is likely to depend on n, the number of observations being handled by the simulation, as <xref ref-type="fig" rid="F3">Figure 3</xref> suggests that choosing k too low means that the final configuration will be underdetermined. Arguments based on stochastic-block models suggest that there is a minimum value of k below which there is inadequate information to reconstruct the true underlying class-membership; however, this values grows only slowly, as &#x2126; (log n) [<xref ref-type="bibr" rid="B19">19</xref>].</p>
<p>With respect to other common dimensional-reduction methods, our &#x201c;spring-like&#x201d; approach with stochastic initialization and gradient descent is related to such familiar approaches like t-SNE [<xref ref-type="bibr" rid="B20">20</xref>], though we do not assign neighbors using a Gaussian or t-distribution but rather use repeated application of the MST algorithm itself. One potential future issue is that the k-MST approach is likely to be sensitive to class imbalance. If one class comes to be vastly outnumbered by another its members may be less likely to be connected in the k-MST, and hence will not experience the attractive forces that produce strong clustering; conversely, the attraction between relatively few similar nodes may be overwhelmed by the attractive force of a much larger number of adjoining, yet dissimilar nodes.</p>
<p>Even in this case, however, we believe it is likely that the members of the less-populated class will tend either to have a higher chance of being connected through the k-MST (by a similar reasoning to that which motivates the MVR test itself), or will form part of a larger region containing &#x201c;similar&#x201d; nodes of the other class (as we see with the considerable number of WT patients whose nodes consistently segregate into the IDH-dominated region). One possible solution to this potential limitation, inspired by work for the MVR test [<xref ref-type="bibr" rid="B21">21</xref>], might be simply to weight the attractive forces within the k-MST simulation in inverse proportion to the number of nodes in the class, so that &#x201c;rarer&#x201d; nodes attract each other most strongly.</p>
<p>One potential computational limitation of our approach relates to its dependence on the creation of MST, which requires creation of a full-graph distance matrix. Since this contains n<sup>2</sup> pairwise distances, the computational overhead will increase with O (n<sup>2</sup>), limiting the number of data-points that can be calculated. However, clustering-based approaches are known which can be used to generate approximate MSTs that run in O (n<sup>3/2</sup>), substantially speeding up the distance-matrix bottleneck [<xref ref-type="bibr" rid="B22">22</xref>]. A natural next step, therefore, will be to implement and evaluate approximate MSTs, which should allow the processing of hundreds of thousands of data-points in reasonable time.</p>
<p>It is well-known that radiomics approaches can be vulnerable to false positives, particularly in the case where there are more radiomic features than there are patients [<xref ref-type="bibr" rid="B23">23</xref>]. Furthermore, it has been noted that &#x201c;choice of the classification model could lead to variations in the predictive values of the radiomic features up to &#x3e;30%&#x201d; [<xref ref-type="bibr" rid="B23">23</xref>]. While our approach largely avoids feature-selection issues by effectively condensing the data into a higher-level statistical, graph-theoretical, indeed structural question, further validation studies on other types of radiomics and imaging data are clearly indicated, as such studies can help to eliminate false positives [<xref ref-type="bibr" rid="B23">23</xref>].</p>
<p>Even taking this into account, we believe the centrality/MVR/k-MST force-directed combination approach presented here has the potential to greatly simplify the analysis of radiomics data, while simultaneously rendering it far more readily interpretable. By relying on the simple, MVR test&#x2014;which is parameter-free except with respect to the variable C, itself derived from the pooled data MST--we avoid numerous somewhat arbitrary aspects of testing and analysis with high-dimensional data. Since the MST itself does not depend on any arbitrary parameters, this too provides a simpler, possibly more &#x201c;objective&#x201d; approach. As raised by a reviewer, there is an important theoretical question, which involves our centrality based analysis. It is based on the MST, and the relationship between MST-based centrality with the original data-based graph centrality remains an open problem.</p>
<p>In conclusion, we have developed a combination of graph-theoretical approaches that provide rapid visualization, significance testing, and dimensionality reduction for very high-dimensional radiomics (and other) datasets, with the potential for considerable streamlining of the workflow and improved &#x201c;explainability&#x201d;. Future investigations will help gauge the effectiveness of this general approach to other radiomics use-cases, as well as to other high-dimensional medical data.</p>
</sec>
</body>
<back>
<sec id="s5">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="sec" rid="s10">Supplementary Materials</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s6">
<title>Author Contributions</title>
<p>OS and DG formulated the design of the study. OS developed and implemented the methodology. All authors provided feedback on its implementation. Y-SC provided the data for the study. OS wrote the first draft, and all authors contributed to writing of subsequent drafts of the manuscript. All authors approved the final version of the manuscript.</p>
</sec>
<sec id="s7">
<title>Funding</title>
<p>This research has been partially supported by the National Cancer Institute through Grant R01 CA129102. The funding agency played no role in the design or analysis of the study.</p>
</sec>
<sec sec-type="COI-statement" id="s8">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ack>
<p>OS and DG would like to acknowledge the support of NCI R01 CA129102. DG acknowledges the support of the Grohne-Stepp Endowment from the University of Colorado Cancer Center.</p>
</ack>
<sec id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fphy.2022.783765/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fphy.2022.783765/full&#x23;supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.PDF" id="SM1" mimetype="application/PDF" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>De Bruijne</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Machine Learning Approaches in Medical Image Analysis: From Detection to Diagnosis</article-title>. <source>Med Image Anal</source> (<year>2016</year>) <volume>33</volume>:<fpage>94</fpage>&#x2013;<lpage>7</lpage>. <pub-id pub-id-type="doi">10.1016/j.media.2016.06.032</pub-id> </citation>
</ref>
<ref id="B2">
<label>2.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bhandari</surname>
<given-names>AP</given-names>
</name>
<name>
<surname>Liong</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Koppen</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Murthy</surname>
<given-names>SV</given-names>
</name>
<name>
<surname>Lasocki</surname>
<given-names>A</given-names>
</name>
</person-group>. <article-title>Noninvasive Determination of IDH and 1p19q Status of Lower-Grade Gliomas Using MRI Radiomics: a Systematic Review</article-title>. <source>AJNR Am J Neuroradiol</source> (<year>2021</year>) <volume>42</volume>(<issue>1</issue>):<fpage>94</fpage>&#x2013;<lpage>101</lpage>. <pub-id pub-id-type="doi">10.3174/ajnr.a6875</pub-id> </citation>
</ref>
<ref id="B3">
<label>3.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calabrese</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Villanueva-Meyer</surname>
<given-names>JE</given-names>
</name>
<name>
<surname>Cha</surname>
<given-names>S</given-names>
</name>
</person-group>. <article-title>A Fully Automated Artificial Intelligence Method for Non-invasive, Imaging-Based Identification of Genetic Alterations in Glioblastomas</article-title>. <source>Sci Rep</source> (<year>2020</year>) <volume>10</volume>(<issue>1</issue>):<fpage>11852</fpage>&#x2013;<lpage>11</lpage>. <pub-id pub-id-type="doi">10.1038/s41598-020-68857-8</pub-id> </citation>
</ref>
<ref id="B4">
<label>4.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname>
<given-names>YS</given-names>
</name>
<name>
<surname>Bae</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Chang</surname>
<given-names>JH</given-names>
</name>
<name>
<surname>Kang</surname>
<given-names>S-G</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>SH</given-names>
</name>
<name>
<surname>Kim</surname>
<given-names>J</given-names>
</name>
<etal/>
</person-group> <article-title>Fully Automated Hybrid Approach to Predict the IDH Mutation Status of Gliomas via Deep Learning and Radiomics</article-title>. <source>Neuro-Oncol.</source> (<year>2021</year>) <volume>23</volume>(<issue>2</issue>):<fpage>304</fpage>&#x2013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1093/neuonc/noaa177</pub-id> </citation>
</ref>
<ref id="B5">
<label>5.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rogers</surname>
<given-names>W</given-names>
</name>
<name>
<surname>Thulasi Seetha</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Refaee</surname>
<given-names>TAG</given-names>
</name>
<name>
<surname>Lieverse</surname>
<given-names>RIY</given-names>
</name>
<name>
<surname>Granzier</surname>
<given-names>RWY</given-names>
</name>
<name>
<surname>Ibrahim</surname>
<given-names>A</given-names>
</name>
<etal/>
</person-group> <article-title>Radiomics: from Qualitative to Quantitative Imaging</article-title>. <source>Bjr</source> (<year>2020</year>) <volume>93</volume>:<fpage>20190948</fpage>. <pub-id pub-id-type="doi">10.1259/bjr.20190948</pub-id> </citation>
</ref>
<ref id="B6">
<label>6.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hosny</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Aerts</surname>
<given-names>HJ</given-names>
</name>
<name>
<surname>Mak</surname>
<given-names>RH</given-names>
</name>
</person-group>. <article-title>Handcrafted versus Deep Learning Radiomics for Prediction of Cancer Therapy Response</article-title>. <source>Lancet Digit Heealth</source> (<year>2019</year>) <volume>1</volume>(<issue>3</issue>):<fpage>E106</fpage>&#x2013;<lpage>107</lpage>. <pub-id pub-id-type="doi">10.1016/S2589-7500(19)30062-7</pub-id> </citation>
</ref>
<ref id="B7">
<label>7.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Arpit</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Jastrz&#x119;bski</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Ballas</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Krueger</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Bengio</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Kanwal</surname>
<given-names>MS</given-names>
</name>
<etal/>
</person-group> <source>A Closer Look at Memorization in Deep Networks</source>. <publisher-name>PMLR</publisher-name> (<year>2017</year>). <fpage>233</fpage>&#x2013;<lpage>42</lpage>. </citation>
</ref>
<ref id="B8">
<label>8.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gower</surname>
<given-names>JC</given-names>
</name>
<name>
<surname>Ross</surname>
<given-names>GJS</given-names>
</name>
</person-group>. <article-title>Minimum Spanning Trees and Single Linkage Cluster Analysis</article-title>. <source>Appl Stat</source> (<year>1969</year>) <volume>18</volume>(<issue>1</issue>):<fpage>54</fpage>&#x2013;<lpage>64</lpage>. <pub-id pub-id-type="doi">10.2307/2346439</pub-id> </citation>
</ref>
<ref id="B9">
<label>9.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Louf</surname>
<given-names>R</given-names>
</name>
<name>
<surname>Jensen</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Barthelemy</surname>
<given-names>M</given-names>
</name>
</person-group>. <article-title>Emergence of Hierarchy in Cost-Driven Growth of Spatial Networks</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A.</source> (<year>2013</year>) <volume>110</volume>(<issue>22</issue>):<fpage>8824</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1222441110</pub-id> </citation>
</ref>
<ref id="B10">
<label>10.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stam</surname>
<given-names>CJ</given-names>
</name>
<name>
<surname>Tewarie</surname>
<given-names>P</given-names>
</name>
<name>
<surname>Van Dellen</surname>
<given-names>E</given-names>
</name>
<name>
<surname>Van Straaten</surname>
<given-names>ECW</given-names>
</name>
<name>
<surname>Hillebrand</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Van Mieghem</surname>
<given-names>P</given-names>
</name>
</person-group>. <article-title>The Trees and the Forest: Characterization of Complex Brain Networks with Minimum Spanning Trees</article-title>. <source>Int J Psychophysiol</source> (<year>2014</year>) <volume>92</volume>(<issue>3</issue>):<fpage>129</fpage>&#x2013;<lpage>38</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijpsycho.2014.04.001</pub-id> </citation>
</ref>
<ref id="B11">
<label>11.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Freeman</surname>
<given-names>LC</given-names>
</name>
</person-group>. <article-title>A Set of Measures of Centrality Based on Betweenness</article-title>. <source>Sociometry</source> (<year>1977</year>) <volume>40</volume>:<fpage>35</fpage>&#x2013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.2307/3033543</pub-id> </citation>
</ref>
<ref id="B12">
<label>12.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Friedman</surname>
<given-names>JH</given-names>
</name>
</person-group>. <article-title>A New Graph-Based Two-Sample Test for Multivariate and Object Data</article-title>. <source>J Am Stat Assoc</source> (<year>2017</year>) <volume>112</volume>(<issue>517</issue>):<fpage>397</fpage>&#x2013;<lpage>409</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.2016.1147356</pub-id> </citation>
</ref>
<ref id="B13">
<label>13.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname>
<given-names>JH</given-names>
</name>
<name>
<surname>Rafsky</surname>
<given-names>LC</given-names>
</name>
</person-group>. <article-title>Multivariate Generalizations of the Wald-Wolfowitz and Smirnov Two-Sample Tests</article-title>. <source>Ann Stat</source> (<year>1979</year>) <fpage>697</fpage>&#x2013;<lpage>717</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1176344722</pub-id> </citation>
</ref>
<ref id="B14">
<label>14.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Friedman</surname>
<given-names>JH</given-names>
</name>
<name>
<surname>Rafsky</surname>
<given-names>LC</given-names>
</name>
</person-group>. <article-title>Graph-theoretic Measures of Multivariate Association and Prediction</article-title>. <source>Ann Stat</source> (<year>1983</year>) <fpage>377</fpage>&#x2013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1176346148</pub-id> </citation>
</ref>
<ref id="B15">
<label>15.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van Griethuysen</surname>
<given-names>JJM</given-names>
</name>
<name>
<surname>Fedorov</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Parmar</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Hosny</surname>
<given-names>A</given-names>
</name>
<name>
<surname>Aucoin</surname>
<given-names>N</given-names>
</name>
<name>
<surname>Narayan</surname>
<given-names>V</given-names>
</name>
<etal/>
</person-group> <article-title>Computational Radiomics System to Decode the Radiographic Phenotype</article-title>. <source>Cancer Res</source>(<year>2017</year>) <volume>77</volume>(<issue>21</issue>):<fpage>e104</fpage>&#x2013;<lpage>e107</lpage>. <pub-id pub-id-type="doi">10.1158/0008-5472.can-17-0339</pub-id> </citation>
</ref>
<ref id="B16">
<label>16.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Newman</surname>
<given-names>MEJ</given-names>
</name>
</person-group>. <article-title>Mathematics of Networks</article-title>. <source>New Palgrave Encycl Econ</source> (<year>2008</year>) <volume>2</volume>:<fpage>1</fpage>&#x2013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.1057/978-1-349-95121-5_2565-1</pub-id> </citation>
</ref>
<ref id="B17">
<label>17.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>van de Haar</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Canisius</surname>
<given-names>S</given-names>
</name>
<name>
<surname>Yu</surname>
<given-names>MK</given-names>
</name>
<name>
<surname>Voest</surname>
<given-names>EE</given-names>
</name>
<name>
<surname>Wessels</surname>
<given-names>LFA</given-names>
</name>
<name>
<surname>Ideker</surname>
<given-names>T</given-names>
</name>
</person-group>. <article-title>Identifying Epistasis in Cancer Genomes: a Delicate Affair</article-title>. <source>Cell</source> (<year>2019</year>) <volume>177</volume>(<issue>6</issue>):<fpage>1375</fpage>&#x2013;<lpage>83</lpage>. <pub-id pub-id-type="doi">10.1016/j.cell.2019.05.005</pub-id> </citation>
</ref>
<ref id="B18">
<label>18.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Barnes</surname>
<given-names>J</given-names>
</name>
<name>
<surname>Hut</surname>
<given-names>P</given-names>
</name>
</person-group>. <article-title>A Hierarchical O(N Log N) Force-Calculation Algorithm</article-title>. <source>nature</source> (<year>1986</year>) <volume>324</volume>(<issue>6096</issue>):<fpage>446</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1038/324446a0</pub-id> </citation>
</ref>
<ref id="B19">
<label>19.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Moore</surname>
<given-names>C</given-names>
</name>
</person-group>. <source>The Computer Science and Physics of Community Detection: Landscapes, Phase Transitions, and Hardness</source>. <publisher-name>ArXiv Prepr ArXiv170200467</publisher-name> (<year>2017</year>). </citation>
</ref>
<ref id="B20">
<label>20.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Van der Maaten</surname>
<given-names>L</given-names>
</name>
<name>
<surname>Hinton</surname>
<given-names>G</given-names>
</name>
</person-group>. <article-title>Visualizing Data Using T-SNE</article-title>. <source>J Mach Learn Res</source> (<year>2008</year>) <volume>9</volume>(<issue>11</issue>). </citation>
</ref>
<ref id="B21">
<label>21.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname>
<given-names>H</given-names>
</name>
<name>
<surname>Chen</surname>
<given-names>X</given-names>
</name>
<name>
<surname>Su</surname>
<given-names>Y</given-names>
</name>
</person-group>. <article-title>A Weighted Edge-Count Two-Sample Test for Multivariate and Object Data</article-title>. <source>J Am Stat Assoc</source> (<year>2018</year>) <volume>113</volume>(<issue>523</issue>):<fpage>1146</fpage>&#x2013;<lpage>55</lpage>. <pub-id pub-id-type="doi">10.1080/01621459.2017.1307757</pub-id> </citation>
</ref>
<ref id="B22">
<label>22.</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhong</surname>
<given-names>C</given-names>
</name>
<name>
<surname>Malinen</surname>
<given-names>M</given-names>
</name>
<name>
<surname>Miao</surname>
<given-names>D</given-names>
</name>
<name>
<surname>Fr&#xe4;nti</surname>
<given-names>P</given-names>
</name>
</person-group>. <source>Fast Approximate Minimum Spanning Tree Algorithm Based on K-Means</source>. <publisher-name>Springer</publisher-name> (<year>2013</year>). p. <fpage>262</fpage>&#x2013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-642-40261-6_31</pub-id>
<article-title>Fast Approximate Minimum Spanning Tree Algorithm Based on K-Means</article-title> </citation>
</ref>
<ref id="B23">
<label>23.</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yip</surname>
<given-names>SSF</given-names>
</name>
<name>
<surname>Aerts</surname>
<given-names>HJWL</given-names>
</name>
</person-group>. <article-title>Applications and Limitations of Radiomics</article-title>. <source>Phys. Med. Biol.</source> (<year>2016</year>) <volume>61</volume>(<issue>13</issue>):<fpage>R150</fpage>&#x2013;<lpage>R166</lpage>. <pub-id pub-id-type="doi">10.1088/0031-9155/61/13/r150</pub-id> </citation>
</ref>
</ref-list>
</back>
</article>