<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neuroinform.</journal-id>
<journal-title>Frontiers in Neuroinformatics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neuroinform.</abbrev-journal-title>
<issn pub-type="epub">1662-5196</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fninf.2023.1266713</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Technology and Code</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>oFVSD: a Python package of optimized forward variable selection decoder for high-dimensional neuroimaging data</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author"><name><surname>Dang</surname> <given-names>Tung</given-names></name><xref rid="aff1" ref-type="aff"><sup>1</sup></xref><xref rid="aff2" ref-type="aff"><sup>2</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/2327753/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Fermin</surname> <given-names>Alan S. R.</given-names></name><xref rid="aff1" ref-type="aff"><sup>1</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/26221/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Machizawa</surname> <given-names>Maro G.</given-names></name><xref rid="aff1" ref-type="aff"><sup>1</sup></xref><xref rid="c001" ref-type="corresp"><sup>&#x002A;</sup></xref><uri xlink:href="https://loop.frontiersin.org/people/41958/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Center for Brain, Mind, and KANSEI Sciences Research, Hiroshima University</institution>, <addr-line>Hiroshima</addr-line>, <country>Japan</country></aff>
<aff id="aff2"><sup>2</sup><institution>Graduate School of Agricultural and Life Sciences, The University of Tokyo</institution>, <addr-line>Tokyo</addr-line>, <country>Japan</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0003">
<p>Edited by: Farouk Nathoo, University of Victoria, Canada</p>
</fn>
<fn fn-type="edited-by" id="fn0004">
<p>Reviewed by: Shailesh Appukuttan, UMR9197 Institut des Neurosciences Paris Saclay (Neuro-PSI), France; Chao Huang, Florida State University, United States</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Maro G. Machizawa, <email>machizawa@hiroshima-u.ac.jp</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>26</day>
<month>09</month>
<year>2023</year>
</pub-date>
<pub-date pub-type="collection">
<year>2023</year>
</pub-date>
<volume>17</volume>
<elocation-id>1266713</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>07</month>
<year>2023</year>
</date>
<date date-type="accepted">
<day>08</day>
<month>09</month>
<year>2023</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2023 Dang, Fermin and Machizawa.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Dang, Fermin and Machizawa</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>The complexity and high dimensionality of neuroimaging data pose problems for decoding information with machine learning (ML) models because the number of features is often much larger than the number of observations. Feature selection is one of the crucial steps for determining meaningful target features in decoding; however, optimizing the feature selection from such high-dimensional neuroimaging data has been challenging using conventional ML models. Here, we introduce an efficient and high-performance decoding package incorporating a forward variable selection (FVS) algorithm and hyper-parameter optimization that automatically identifies the best feature pairs for both classification and regression models, where a total of 18 ML models are implemented by default. First, the FVS algorithm evaluates the goodness-of-fit across different models using the k-fold cross-validation step that identifies the best subset of features based on a predefined criterion for each model. Next, the hyperparameters of each ML model are optimized at each forward iteration. Final outputs highlight an optimized number of selected features (brain regions of interest) for each model with its accuracy. Furthermore, the toolbox can be executed in a parallel environment for efficient computation on a typical personal computer. With the optimized forward variable selection decoder (oFVSD) pipeline, we verified the effectiveness of decoding sex classification and age range regression on 1,113 structural magnetic resonance imaging (MRI) datasets. Compared to ML models without the FVS algorithm and with the Boruta algorithm as a variable selection counterpart, we demonstrate that the oFVSD significantly outperformed across all of the ML models over the counterpart models without FVS (approximately 0.20 increase in correlation coefficient, <italic>r</italic>, with regression models and 8% increase in classification models on average) and with Boruta variable selection algorithm (approximately 0.07 improvement in regression and 4% in classification models). Furthermore, we confirmed the use of parallel computation considerably reduced the computational burden for the high-dimensional MRI data. Altogether, the oFVSD toolbox efficiently and effectively improves the performance of both classification and regression ML models, providing a use case example on MRI datasets. With its flexibility, oFVSD has the potential for many other modalities in neuroimaging. This open-source and freely available Python package makes it a valuable toolbox for research communities seeking improved decoding accuracy.</p>
</abstract>
<kwd-group>
<kwd>machine learning</kwd>
<kwd>forward variable selection</kwd>
<kwd>optimized hyperparameter</kwd>
<kwd>neural decoding</kwd>
<kwd>MRI</kwd>
<kwd>VBM (voxel-based morphometry)</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="78"/>
<page-count count="14"/>
<word-count count="11334"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Front. Neuroinform</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1.</label>
<title>Introduction</title>
<p>Neuroimaging data such as structural and functional magnetic resonance imaging (MRI) data provide information about the functional neuroanatomy with a high spatial resolution and play an essential role in providing researchers with unprecedented access to the inner workings of the brain of a healthy individual or an individual with a neurological disease or psychiatric disorder (<xref ref-type="bibr" rid="ref77">Zhu et al., 2019</xref>). Identifying brain regions that differentiate healthy and nonhealthy participants (<xref ref-type="bibr" rid="ref51">Nielsen et al., 2020</xref>) or the prognosis of patients&#x2019; pathological states (<xref ref-type="bibr" rid="ref35">Janssen et al., 2018</xref>) is essential for neuroscientific studies. Needless to say, the impact of those findings is rooted in the accuracy of their decoding.</p>
<p>In the last couple of decades, we have seen a surge of studies turning to ML models to extract exciting new information from neuroimaging data. For example, the partial least squares (PLS) model was proposed to extract distributed neural signal changes by taking advantage of image elements&#x2019; spatial and temporal dependencies (<xref ref-type="bibr" rid="ref45">McIntosh et al., 1996</xref>; <xref ref-type="bibr" rid="ref46">McIntosh and Lobaugh, 2004</xref>). The adaptive boosting model (&#x2018;AdaBoost&#x2019;) was proposed to classify addiction disorder patients and healthy controls based on observed 3-dimensional functional brain images (<xref ref-type="bibr" rid="ref68">Warren and Moustafa, 2022</xref>). The existence of diverse ML models has raised the question of which and when an ML model is better suited to extract important new information from neuroimaging data (<xref ref-type="bibr" rid="ref52">O&#x2019;Toole et al., 2007</xref>; <xref ref-type="bibr" rid="ref55">Pereira et al., 2009</xref>). However, the selection of the most appropriate models for a specific dataset and purposes is challenging for people with little experience in ML since the appropriate choice of a model depends on the number of features (<xref ref-type="bibr" rid="ref59">Saeys et al., 2007</xref>).</p>
<p>The curse of dimensionality of neuroimaging data can negatively affect the generalization performance of ML models, leading to estimation instability, model overfitting, local convergence, and large estimation errors (<xref ref-type="bibr" rid="ref49">Mwangi et al., 2014</xref>). For example, the decoding abilities of models that depend on specific distributions of data, such as geometric distributions of data, can be significantly influenced in the high-dimensional data space. A na&#x00EF;ve learning model requires the number of training data points to be an exponential function of the attribute dimension (<xref ref-type="bibr" rid="ref34">Jain et al., 2000</xref>). Furthermore, the problem of the high dimensionality of neuroimaging data (e.g., an extremely large number of voxels in fMRI research) poses a number of challenges (<xref ref-type="bibr" rid="ref67">Vul et al., 2009</xref>) even if a model is based on nonparametric strategies (<xref ref-type="bibr" rid="ref32">Huang et al., 2012</xref>). For example, in the random forest model, available features are randomly sampled to generate different subspaces of features used to train each decision tree in an ensemble (<xref ref-type="bibr" rid="ref39">Kuncheva et al., 2010</xref>). Because it is typical to observe only a few features out of many that significantly contribute to the performance of these models, a large number of irrelevant features appear in these subspaces. Thus, the average strength of decision trees can be diluted, thereby increasing the generalization error of the random forest model. These problems also exist in neural network models when the high dimensionality of data confounds learning techniques, and the network must allocate its resources to represent many irrelevant components (<xref ref-type="bibr" rid="ref62">Scott, 1992</xref>).</p>
<p>Recent advancements in neuroimaging technologies have also increased the data size; namely, the total number of features to be considered has increased. Therefore, a feature reduction technique has become one of the essential aspects of neuroimaging research (<xref ref-type="bibr" rid="ref49">Mwangi et al., 2014</xref>). The size of neuroimaging data may lead to a computational burden. However, building a pipeline for hundreds of thousands of brain regions can be very costly and time-consuming. Recently, several ML methods incorporating parallel computing environments have been developed (<xref ref-type="bibr" rid="ref72">Xing et al., 2016</xref>); implementing a fast and efficient pipeline would be of potential application for analyzing a large amount of neuroimaging data.</p>
<p>While many algorithms have been proposed, one thorough ML model is forward variable selection (FVS). The FVS algorithm is a member of the stepwise feature selection algorithm family (<xref ref-type="bibr" rid="ref29">Guyon and Elisseeff, 2003</xref>; <xref ref-type="bibr" rid="ref41">Kutner et al., 2005</xref>; <xref ref-type="bibr" rid="ref70">Weisberg, 2005</xref>; <xref ref-type="bibr" rid="ref10">Chandrashekar and Sahin, 2014</xref>). It is also one of the first and most popular algorithms for causal feature selection in some fields, such as gene selection, microarray data analysis, and gene expression data analysis (<xref ref-type="bibr" rid="ref53">Ooi and Tan, 2003</xref>; <xref ref-type="bibr" rid="ref5">Blanco et al., 2004</xref>; <xref ref-type="bibr" rid="ref36">Jirapech-Umpai and Aitken, 2005</xref>; <xref ref-type="bibr" rid="ref59">Saeys et al., 2007</xref>). The powerful nature of feature decoding in the analysis of high-dimensional microbiome data has also been demonstrated (<xref ref-type="bibr" rid="ref15">Dang et al., 2022</xref>; <xref ref-type="bibr" rid="ref14">Dang and Kishino, 2022</xref>). The FVS can be a powerful additional tool for neuroimaging research.</p>
<p>Despite a significant rise in the application of ML models and their potential contributions to understanding brain functions, neuroimaging data are ill-posed to the high-dimensionality problem. Here, we propose a state-of-the-art and effective ML package as a solution to the high-dimensionality problem of neuroimaging data that is easy to use by neuroscientists interested in applying ML models to decode their neuroimaging data with little computational programming. In this study, we developed a novel decoding pipeline to overcome these challenges by combining two frameworks. First, we developed an ML framework incorporating an FVS algorithm that integrates model selection steps to detect the minimal set of features that could maximize the predictive performance. Second, the pipeline selects the best model from a predetermined set of regression and classifier models. This simple yet comprehensive two-stage algorithm automatically and effectively identifies important features from neuroimaging data. Moreover, because the nature of the FVS is computationally intensive and time-costly, the toolbox executes in a parallel environment to save computational costs. As a proof of concept of our approach to neuroimaging datasets, structural neuroimaging data were acquired to examine the feasibility of our proposed FVS toolbox to decode the neuroanatomical representation of (1) biological sex and (2) age with binary classification and multiple regression models, respectively.</p>
</sec>
<sec sec-type="Methods|material" id="sec2">
<label>2.</label>
<title>Methods and material</title>
<sec id="sec3">
<label>2.1.</label>
<title>Forward variable selection (FVS) algorithm</title>
<p>The FVS algorithm requires an ML model for feature selection and uses its performance to evaluate and determine which features are selected. The key idea behind the FVS algorithm is to select a feature that provides the largest improvement in terms of the predictive performance of the ML model and append this feature to the set of selected features in each forward iteration. The iterations stop when there is no feature improvement in the performance upon adding a feature or the maximum number of selected features has been reached.</p>
<p>In this proof of concept to decode either sex or age from regional gray matter volume, we used the FVS algorithm to identify a small number of features (i.e., regions of interest) to improve the performance of ML models in the subsequent step of regression or classification. At each learning step, a brain region that provides the largest increase in the predictive performance of regression or classification models is selected and added to the set of selected brain regions. This process continues until there is no further performance improvement after selecting a brain region or the maximum number of selected brain regions that have been set <italic>a priori</italic> has been reached. To save computational time or to restrict the number of to-be-selected features for a certain purpose, users should specify a maximum number of brain regions for the FVS algorithm. If not, a maximum value is all the available brain regions in the data (246 ROIs). In this study, 100 ROIs is the maximum number of brain regions for the FVS algorithm. Model selection for brain region signature identification can also be performed using the FVS algorithm (see <xref rid="fig1" ref-type="fig">Figure 1</xref>). At each forward iteration, given the selected brain regions, samples were randomly split into a training dataset comprising 70% of the samples and a test dataset comprising the remaining 30% of the samples. To select an appropriate model configuration for a specific task, such as a prediction of age, fivefold cross-validation was performed on the training data to optimize the respective hyperparameters. Two standard algorithms of hyperparameter optimization, the grid search and random search strategies with cross-validation, were implemented to select the best values for the parameters of the ML model (<xref ref-type="bibr" rid="ref4">Bisong, 2019</xref>; <xref ref-type="bibr" rid="ref1">Agrawal, 2021</xref>). The best-performing hyperparameters for each model were achieved when the MSE was minimized.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Workflow schematics of the automatic ML toolbox coupled with the forward variable selection (FVS) algorithm. All features, i.e., the gray matter volume data from each region of interest (ROI), undergo the FVS step, followed by either regression-based or classification-based ML with K-fold cross-validation (CV). The random search and grid search strategies with cross-validation were adopted to optimize the hyperparameters of the ML models at each iteration of the FVS algorithm. The final outcomes were evaluated based on the MSE and MAE for regression-based models and the AUC and confusion matrix for classification-based model. <italic>n</italic> is the number of samples, m is the total number of ROIs (246 ROIs in this study) and x is the number of ROIs that the user wants to select.</p>
</caption>
<graphic xlink:href="fninf-17-1266713-g001.tif"/>
</fig>
<p>Based on the specific numbers of parameters of the ML model, grid search with cross-validation, all parameter combinations are exhaustively considered, while with the random search with cross-validation, a given number of values are randomly selected from a parameter space was considered for parameter optimization (<xref ref-type="bibr" rid="ref3">Bergstra and Bengio, 2012</xref>). These search strategies suffer from high-dimensional spaces but can often easily be parallelized since the hyperparameter values that the algorithm works with are usually independent of each other. Therefore, the ML models have a large number of parameters, such as the random forest and decision tree models. The random search with a cross-validation strategy is used to balance computational time and predictive accuracy. The grid search with cross-validation strategy is used for ML models with a few parameters, such as the lasso or ridge models. Following the hyperparameter optimization, the best ML model is specified, as well as the final selected features, namely brain regions in this case.</p>
<p>The FVS algorithm is implemented in a parallel computing environment to reduce the computational burden in terms of time cost. A number of packages provide high-performance computing solutions in Python (<xref ref-type="bibr" rid="ref54">Palach, 2014</xref>). We used the thread-based parallelism and process-based parallelism that is provided in the joblib package to separate Python worker processes to execute tasks on separate CPUs. To parallelize each FVS iteration, the input variables were separated randomly into a number of subsets. Because of the high dimensionality of neuroimaging data, the number of these subsets (or comparisons) is usually larger than the number of processors in a single computer system. In a previous study, a computer-friendly procedure was proposed for very high-dimensional microbiome data (<xref ref-type="bibr" rid="ref14">Dang and Kishino, 2022</xref>). To introduce efficient computation, queues were created to randomly assign subsets to each processor that runs the computational processes from its own privately prepared queue (<xref ref-type="bibr" rid="ref14">Dang and Kishino, 2022</xref>).</p>
<p>As the counterpart feature selection method we propose, the Boruta algorithm (<xref ref-type="bibr" rid="ref40">Kursa and Rudnicki, 2010</xref>) was tested to compare the performance of the FVS. <xref ref-type="bibr" rid="ref40">Kursa and Rudnicki (2010)</xref> originally developed the wrapper algorithm to identify all important variables within a classification framework. The Boruta feature selection algorithm is applied in bioinformatics areas to select protein targets (<xref ref-type="bibr" rid="ref56">Pietzner et al., 2021</xref>; <xref ref-type="bibr" rid="ref2">Al-Nesf et al., 2022</xref>), microbial functions (<xref ref-type="bibr" rid="ref16">Diamond et al., 2019</xref>; <xref ref-type="bibr" rid="ref60">Saffouri et al., 2019</xref>; <xref ref-type="bibr" rid="ref18">Edwinson et al., 2022</xref>), and metabolomic profiles (<xref ref-type="bibr" rid="ref47">Metwaly et al., 2020</xref>; <xref ref-type="bibr" rid="ref43">Mayneris-Perxachs et al., 2022</xref>). The main idea of the Boruta algorithm is to create shadow features by randomly permuting the values of each original feature. This permutation is to generate a null distribution that represents the expected importance scores of features. Then, the original features and their corresponding shadow features are used to train the random forest classifier. The importance of each original and shadow feature is determined based on the random forest model. The z-score of the original feature is then computed by comparing its importance score with the distribution of importance scores of the corresponding shadow features. If the z-score is significantly higher than the expected chance level, it indicates that the original feature is more important than the shadow features. In this study, the maximum value is all the available brain regions in the data (246 ROIs) for the Boruta algorithm.</p>
</sec>
<sec id="sec4">
<label>2.2.</label>
<title>Regression-based ML models</title>
<p>We explored various ML models to examine how a regional brain structure could contain information representing age. After the preprocessing of structural imaging data, the input (target features) included high-dimensional structural gray matter volumetric data from 1,113 samples in 246 brain regions (see Section 3 for details). We applied a variety of ML models to identify features and to reduce the dimension of this input with parametric regularization models for feature selection and nonparametric models that perform a random sampling of the available features to generate different subspaces of features to achieve a trade-off between bias and variance. In our toolbox, we selected two nonparametric models and 10 parametric models.</p>
<p>For the nonparametric regression models, the commonly used decision tree regression, random forest (RF), and Gaussian process models were selected (<xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). (1) Decision tree regression is a supervised learning model that sets up a decision rule depending on the features at every interior node (<xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). The features selected for the first partition at the root have the largest relevance. This feature selection procedure is recursively repeated for each subset at the node until further partitioning becomes impossible. The decision tree regression is typically considered to analyze MRI images (<xref ref-type="bibr" rid="ref50">Naik and Patel, 2014</xref>; <xref ref-type="bibr" rid="ref23">Filli et al., 2018</xref>; <xref ref-type="bibr" rid="ref38">Kim et al., 2018</xref>). (2) The random forest (RF) is a modification of the bagging regression that aggregates a large collection of decision trees (<xref ref-type="bibr" rid="ref7">Breiman, 2001</xref>). The primary step in building an ensemble of decision trees is to randomly sample the available features to generate different subspaces of features at each node of each unpruned decision tree. Using this strategy, better estimation performances can be obtained compared with using a single decision tree because each tree estimator has a low bias but high variance, whereas the bagging process of RF achieves a bias-variance trade-off. The random forest (RF) model has become a standard data analysis tool in multiple areas, such as bioinformatics (<xref ref-type="bibr" rid="ref6">Boulesteix et al., 2012</xref>; <xref ref-type="bibr" rid="ref22">Ferreira and Figueiredo, 2012</xref>) and neuroimaging analysis (<xref ref-type="bibr" rid="ref48">Mitra et al., 2014</xref>; <xref ref-type="bibr" rid="ref20">Eshaghi et al., 2016</xref>). (3) The Gaussian process (GP) is a nonparametric model that is a natural generalization of a multivariate Gaussian distribution to a Gaussian distribution over a specific family of functions, such as kernel functions (<xref ref-type="bibr" rid="ref57">Rasmussen, 2003</xref>). In GP regression, a prior distribution is proposed directly over the nonlinear function space rather than specifying a parametric family of nonlinear functions. Different kernels can be used to express different structures observed in the data. Thus, the GP has a large degree of flexibility in capturing the underlying signals without imposing strong modeling assumptions. This property makes the GP an attractive model for analyzing genetic data (<xref ref-type="bibr" rid="ref12">Chu et al., 2005</xref>) as well as MRI data (<xref ref-type="bibr" rid="ref69">Wassermann et al., 2010</xref>).</p>
<p>We selected eight parametric ML models, including &#x2018;ridge regression,&#x2019; &#x2018;least absolute shrinkage and selection operator (Lasso) regression,&#x2019; &#x2018;kernel ridge regression,&#x2019; &#x2018;multitask Lasso regression, least angle regression (Lar),&#x2019; &#x2018;LassoLar regression,&#x2019; &#x2018;elastic net regression,&#x2019; and &#x2018;regularized linear model with stochastic gradient descent (SGD).&#x2019; (1) Ridge regression is a linear least squares model that uses L2 regularization or weight decay to control the relative importance of features (<xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). L2 regularization encourages weight values to decay toward zero. Thus, ridge regression can be used to overcome the disadvantages of the ordinary least square method, i.e., the variance in the estimate of the linear transform may be large because the number of features is significantly larger than the number of samples. (2) Least absolute shrinkage and selection operator (Lasso) regression, which is another type of linear regression, uses L1 regularization and can eliminate a number of coefficients from the model by adding a penalty equal to the absolute value of their magnitude (<xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). (3) Kernel ridge regression is an extension of ridge regression that is used when the number of dimensions can be much larger, or even infinitely larger, than the number of samples (<xref ref-type="bibr" rid="ref66">Vovk, 2013</xref>). The main idea is to propose the kernel trick to convert the original data space into the fancy feature space that can significantly reduce the computational burden of learning processes. (4) Multitask Lasso regression generalizes the Lasso to the multitask setting by replacing the L1-norm regularization term with the sup-norm regularization sum (<xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). (5) The least angle regression (Lar) model is the modification of the Lasso and the forward stagewise linear regression models, where the number of features is significantly greater than the number of samples (<xref ref-type="bibr" rid="ref19">Efron et al., 2004</xref>). At each iteration, Lars selects the feature most correlated with the target. If multiple features have a similar correlation, the direction equiangular between the features is moved forward. (6) The Lasso model fit with least angle regression (LassoLar) is the combination of Lar and Lasso and is implemented to improve the variable selection (<xref ref-type="bibr" rid="ref19">Efron et al., 2004</xref>). (7) The elastic net model is the generalization of ridge regression and lasso. This model proposes the elastic net penalty, which controls the coefficients&#x2019; balance between the L1 and L2 regularization (<xref ref-type="bibr" rid="ref78">Zou and Hastie, 2005</xref>; <xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). Thus, an elastic net can be used to perform feature selection in a high-dimensional space. And (8) Regularized linear model with stochastic gradient descent (SGD) learning is an extension of the ridge, Lasso, and elastic net models with a large number of training samples implemented with a plain stochastic gradient descent learning routine (<xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). See the <xref rid="SM1" ref-type="supplementary-material">Supplementary material</xref> for detailed explanations for each ML model.</p>
<p>For each ML model, the parameter optimization steps were developed. Specifically, in the cases of the ridge, Lasso, multitask Lasso, Lar, and LassoLar models, the complexity of the parameters that control the amount of shrinkage was optimized. The elastic net model includes an additional parameter that controls a combination of L1 and L2 penalties separately. The main parameters of the regularized linear model with SGD learning were loss functions, penalty options, and the learning rate schedule, whereas those of the kernel ridge regression model were kernel options that include linear, Laplacian, Gaussian, and sigmoid kernels, the regularization parameter, and the kernel coefficient, and those of the decision tree model were the maximum depth of the tree, the minimum number of samples required to split an internal node, the minimum number of the samples required to be at a leaf node, the function to measure the quality of a split, the strategy used to choose the split at each node, and the number of features to consider when looking for the best split. The parameters of the random forest regression were the same as those of the decision tree regression, except that the number of trees was an additional parameter.</p>
<p>As a criterion to compare the performance of regression, the mean squared error (MSE), mean absolute error (MAE), and Spearman correlation coefficients that were calculated between the predicted and the true values were calculated (<xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). In the field of statistics, the Akaike or Bayesian information criteria (also known as AIC or BIC, respectively) are widely used indices to quantify the fit of a model (<xref ref-type="bibr" rid="ref9">Burnham and Anderson, 2004</xref>); however, these information criteria methods do not apply for nonparametric regression models (e.g., decision tree, random forest). Thus, we selected the MSE and MAE, which are applicable across all models.</p>
</sec>
<sec id="sec5">
<label>2.3.</label>
<title>Classification-based ML models</title>
<p>We also extended the application of our model to study the binary or multi-class classification. We examined how brain structure could contain information representing sex (male or female). The automated classification models include nonparametric models, such as decision tree, random forest, gradient boosting models, extreme gradient boosting, and extremely randomized trees, which have the capability of regression and classification. (1) Decision trees are commonly utilized classification models in various fields, such as machine learning and data mining (<xref ref-type="bibr" rid="ref25">Gavankar and Sawarkar, 2017</xref>). Decision trees include a number of tests or attribute nodes linked to subtrees and decision nodes labeled with a class, i.e., a decision. A sample is classified by starting at the root node of the tree. Each node represents features in a group to be classified, and each subset defines a value that can be taken by the node (<xref ref-type="bibr" rid="ref30">Hastie et al., 2009</xref>). The entropy, Gini index, and information gain are the standard measures of a dataset&#x2019;s impurity or randomness in decision tree classification. (2) Random forest classification is one of the most popular ensemble models that can be used to avoid the tendency of simple decision trees to overfit (<xref ref-type="bibr" rid="ref7">Breiman, 2001</xref>). Similar to regression, random forest classification proposes a slightly randomized training process to build multiple decision trees independently. The randomization processes include using only a random subset of the whole training dataset to build each tree and using a random subset of the features or a random splitting point when considering an optimal split. (3) The gradient boosting model is an ensemble model that uses the boosting technique to combine a sequence of weak decision trees (<xref ref-type="bibr" rid="ref24">Friedman, 2001</xref>). Each tree in the gradient boosting fits the residuals from the previous tree. Thus, the errors of the previous tree are minimized, and the overall accuracy and robustness of the model are considerably improved. (4) Extreme gradient boosting is an efficient and scalable implementation of the gradient boosting model for sparse data with billions of examples (<xref ref-type="bibr" rid="ref11">Chen and Guestrin, 2016</xref>). (5) Extremely randomized trees are another model to improve the performance of decision trees by generating diverse ensembles (<xref ref-type="bibr" rid="ref26">Geurts et al., 2006</xref>). The main idea of this model is to inject randomness into the training process by selecting the best splitting attribute from a random subset of features. However, in contrast to the random forest, the bootstrap instances procedures are implemented by extremely randomized trees. We provide specific explanations for each ML model in the <xref rid="SM1" ref-type="supplementary-material">Supplementary material</xref>.</p>
<p>Because of the computational burdens of nonparametric models, random search strategies with cross-validation are implemented for the parameter optimization steps. The parameters of decision tree classification, such as the maximum depth of the tree and the minimum number of samples required to split an internal node, are similar to those of regression. The Gini index and entropy are used to measure the quality of a split in classification. Moreover, a large number of parameters of random forest, gradient boosting, extreme gradient boosting, and extremely randomized trees are similar to the parameters of decision tree classification. However, several special parameters can significantly influence performance. Specifically, the number of trees is the most important parameter of random forest classification. The necessary parameters of gradient boosting classification include the loss function to binomial and multinomial deviance, the function to measure the quality of a split, the function to measure the quality of a split, and the number of boosting stages. The parameters of extreme gradient boosting are similar to those of gradient boosting. Its computational speed is faster because it has an option for the number of parallel trees constructed during each iteration. An important parameter of extremely randomized trees is the number of trees in the forest, and the bootstrapping technique is not used to build each tree.</p>
<p>Simple parametric models, such as logistic regression and na&#x00EF;ve Bayes, were also included in the classification models. (1) Logistic regression is a standard model for building prediction models for classification. Due to the high-dimensional problems of multiple areas (<xref ref-type="bibr" rid="ref8">B&#x00FC;hlmann and van de Geer, 2011</xref>), ridge and Lasso penalties are added to penalized logistic modeling for the feature selection step. This model has been applied for the analysis of genetic datasets to select a subset of genes that can provide more accurate diagnostic methods (<xref ref-type="bibr" rid="ref42">Liao and Chin, 2007</xref>; <xref ref-type="bibr" rid="ref71">Wu et al., 2009</xref>). (2) Na&#x00EF;ve Bayes is a classification model that refers to constructing a Bayesian probabilistic model to assign a posterior class probability to each sample (<xref ref-type="bibr" rid="ref44">McCallum and Nigam, 1998</xref>). The important assumption of this model is that the features constituting the sample are conditionally independent given the class. The na&#x00EF;ve Bayes model is fast, easy to implement, and relatively effective for the classification of biological datasets (<xref ref-type="bibr" rid="ref74">Yousef et al., 2007</xref>). The grid search strategy with cross-validation is implemented for the parameter optimization steps. The parameter in logistic classification is an elastic net mixing parameter to control the combination of the L1 and L2 regularization. The na&#x00EF;ve Bayes classification parameter is an additive (Laplace/Lidstone) smoothing parameter.</p>
<p>To evaluate the decoding performance, three main criteria were compared across tested models: &#x2018;precision&#x2019; is defined as the number of true positives over the number of true positives plus the number of false positives, &#x2018;recall&#x2019; is defined as the number of true positives over the number of true positives plus the number of false-negatives, and the &#x2018;F1 score&#x2019; is defined as the harmonic mean of precision and recall. <xref rid="fig1" ref-type="fig">Figure 1</xref> depicts the steps for searching for important features using FVS, the application of each ML model, and how these computations are appropriately decomposed in a parallel computation manner. All parallel computations were run on a workstation computer (Intel Xeon Gold 6230 Processor 2.10&#x2009;GHz&#x2009;&#x00D7;&#x2009;2, 40 cores, 2 threads per core, 128 Gb RAM) under Ubuntu 20.04.1 LTS.</p>
</sec>
<sec id="sec6">
<label>2.4.</label>
<title>Neuroimaging data samples</title>
<p>To examine the feasibility of the proposed pipeline in neuroimaging, we acquired high-resolution structural MRI scans of a large number of healthy subjects from the Human Connectome Project (HCP). This dataset includes 1,113 samples. The dataset had four age ranges: 22&#x2013;25, 26&#x2013;30, 31&#x2013;35, and more than 36&#x2009;years; 507 males and 606 females. The structural images were segmented into gray matter, white matter, and cerebrospinal fluid and normalized (1&#x00D7;1&#x00D7;1 voxel size) into a template space using standard parameters implemented in the Computational Neuroanatomy Toolbox (CAT12). During the segmentation process, CAT12 implemented an automated parcellation of the gray matter to extract the gray matter volume in native space from 246 cortical and subcortical brain regions according to neuroanatomical landmarks based on the Brainnetome Atlas<xref rid="fn0001" ref-type="fn"><sup>1</sup></xref> (<xref ref-type="bibr" rid="ref21">Fan et al., 2016</xref>). CAT12 was also used to estimate individual values of the total intracranial volume (TIV), which was included as a covariate of no interest for the classification and regression models. Notably, the pipeline technically works for the whole-brain voxel-based dataset; however, these segmented data were used for simplicity.</p>
<p>Here, we provide a use case example to identify the best model to predict the target variable. More specifically, the gray matter volume data from 246 Brainnetome regions were selected as target features to predict the age and sex of participants using regression and classification models, respectively.</p>
</sec>
<sec id="sec7">
<label>2.5.</label>
<title>Package structure</title>
<p>Our framework includes two core modules: automatic ML models and FVS algorithm for regression and classification. First, ML models and the FVS algorithm were implemented using Python programming to optimize the parallel computations that could significantly reduce the computation time. The scikit-learn library in Python was used to implement core computational techniques for the random forest classifier. Our Python package to implement the proposed model is available on GitHub.<xref rid="fn0002" ref-type="fn"><sup>2</sup></xref> In general, each user creates a short script of regression or classification that contains (1) automatic ML models for the input dataset and (2) the FVS algorithm combined with the best ML model in step (1). For example, the script for regression after controlling the effects of variables such as the total intracranial volume (TIV) is short.</p>
<boxed-text>
<p>&#x003E;&#x003E;&#x003E; from Auto_ML_Regression import AutoML_Regression</p>
<p>&#x003E;&#x003E;&#x003E; from FVS_Regression import AutoML_FVS_Regression</p>
<p>&#x003E;&#x003E;&#x003E; AutoML_Regression.fit(X_train, y_train, X_test, y_test)</p>
</boxed-text>
<p>This function runs 11 ML regression models to select the best model for the input dataset. The output of this function is a table that shows the rank of performances of 11 ML regression models based on their performance.</p>
<boxed-text>
<p>&#x003E;&#x003E;&#x003E; AutoML_FVS_Regression.fit(X_train, y_train, X_test, y_test,</p>
<p>model&#x2009;=&#x2009;&#x201C;LassoLars,&#x201D; n_selected_features&#x2009;=&#x2009;100)</p>
</boxed-text>
<p>After selecting the best ML model, the user implements the function that runs the FVS algorithm to identify an important group of ROIs. For example, the LassoLars model is the best model with the smallest value of MSE in our dataset. Thus, we want to combine the LassoLars model with the FVS algorithm, and the maximum number of features that we want to set is 100. In this case, we define a model as &#x201C;LassoLars&#x201D; and &#x2018;n_selected_features&#x2019; at 100. The details of the parameters and outputs of all functions in our package are provided in the README.md file on GitHub. <xref rid="tab1" ref-type="table">Table 1</xref> shows the main functions of our package.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>An overview of the main functions in the FVSdecoder package.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Function</th>
<th align="left" valign="top">Purpose</th>
<th align="left" valign="top">Output</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top" colspan="3">Functions from AutoML_Regression</td>
</tr>
<tr>
<td align="left" valign="top">fit()</td>
<td align="left" valign="top">Automatic select the best model out of 11 ML regression models</td>
<td align="left" valign="top">A table shows a rank of performances of 11 ML regression</td>
</tr>
<tr>
<td align="left" valign="top">evaluate_regression()</td>
<td align="left" valign="top">Show the performance of ML regression</td>
<td align="left" valign="top">A table shows the MSE and Spearman correlation</td>
</tr>
<tr>
<td align="left" valign="top" colspan="3">Functions from AutoML_Classification</td>
</tr>
<tr>
<td align="left" valign="top">fit()</td>
<td align="left" valign="top">Automatic select the best model out of 9 ML classification models</td>
<td align="left" valign="top">A table shows a rank of performances of 9 ML classification</td>
</tr>
<tr>
<td align="left" valign="top">evaluate_regression()</td>
<td align="left" valign="top">Show the performance of ML classification</td>
<td align="left" valign="top">A table shows accuracy, precision, recall, and F1 score</td>
</tr>
<tr>
<td align="left" valign="top" colspan="3">Functions from AutoML_FVS_Regression</td>
</tr>
<tr>
<td align="left" valign="top">fit()</td>
<td align="left" valign="top">Combine forward variable selection (FVS) with 11 ML regression models</td>
<td align="left" valign="top">A table shows the rank of performances of ML regression for a number of features. A table shows a number of selected features</td>
</tr>
<tr>
<td align="left" valign="top" colspan="3">Functions from AutoML_FVS_Classification</td>
</tr>
<tr>
<td align="left" valign="top">fit()</td>
<td align="left" valign="top">Combine forward variable selection (FVS) with 9 ML classification models</td>
<td align="left" valign="top">A table shows the rank of performances of ML classification for a number of features. A table shows a number of selected features</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec sec-type="results" id="sec8">
<label>3.</label>
<title>Results</title>
<sec id="sec9">
<label>3.1.</label>
<title>Improved accuracy for regression models to predict age</title>
<p><xref rid="tab2" ref-type="table">Table 2</xref> summarizes the MSE values for each ML model with the Boruta algorithm, with and without the FVS algorithm, to predict the age of healthy individuals. For the comparisons without FVS, the best performance and the smallest MSE (MSE&#x2009;=&#x2009;0.4541) were obtained using the LassoLars regression model. MSE values were normally distributed for all variable selection algorithms (Lilliefors corrected Shapiro&#x2013;Wilk test all <italic>p</italic>&#x2009;&#x003E;&#x2009;0.18).</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Accuracies of the ML models as assessed by MSE to predict age.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">MSE without FVS (#ROIs)</th>
<th align="center" valign="top">MSE with Boruta (#ROIs)</th>
<th align="center" valign="top">MSE with FVS (#ROIs)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">LassoLar</td>
<td align="center" valign="top">0.4541 (246)</td>
<td align="center" valign="top">0.4023 (62)</td>
<td align="center" valign="top">0.3686 (54)</td>
</tr>
<tr>
<td align="left" valign="top">Random forest</td>
<td align="center" valign="top">0.4807 (246)</td>
<td align="center" valign="top">0.4156 (75)</td>
<td align="center" valign="top">0.3722 (63)</td>
</tr>
<tr>
<td align="left" valign="top">Gaussian process</td>
<td align="center" valign="top">0.4855 (246)</td>
<td align="center" valign="top">0.4379 (83)</td>
<td align="center" valign="top">0.3895 (78)</td>
</tr>
<tr>
<td align="left" valign="top">Ridge</td>
<td align="center" valign="top">0.4900 (246)</td>
<td align="center" valign="top">0.4418 (79)</td>
<td align="center" valign="top">0.3928 (81)</td>
</tr>
<tr>
<td align="left" valign="top">Elastic net</td>
<td align="center" valign="top">0.4909 (246)</td>
<td align="center" valign="top">0.4653 (77)</td>
<td align="center" valign="top">0.4011 (73)</td>
</tr>
<tr>
<td align="left" valign="top">Lars</td>
<td align="center" valign="top">0.4988 (246)</td>
<td align="center" valign="top">0.4728 (61)</td>
<td align="center" valign="top">0.4171 (68)</td>
</tr>
<tr>
<td align="left" valign="top">Lasso</td>
<td align="center" valign="top">0.5034 (246)</td>
<td align="center" valign="top">0.4831 (67)</td>
<td align="center" valign="top">0.4265 (59)</td>
</tr>
<tr>
<td align="left" valign="top">Kernel ridge</td>
<td align="center" valign="top">0.5061 (246)</td>
<td align="center" valign="top">0.4927 (82)</td>
<td align="center" valign="top">0.4402 (71)</td>
</tr>
<tr>
<td align="left" valign="top">Multitask lasso</td>
<td align="center" valign="top">0.5341 (246)</td>
<td align="center" valign="top">0.5156 (66)</td>
<td align="center" valign="top">0.4578 (52)</td>
</tr>
<tr>
<td align="left" valign="top">Decision tree</td>
<td align="center" valign="top">0.5669 (246)</td>
<td align="center" valign="top">0.5318 (71)</td>
<td align="center" valign="top">0.4669 (76)</td>
</tr>
<tr>
<td align="left" valign="top">Stochastic gradient descent</td>
<td align="center" valign="top">0.5687 (246)</td>
<td align="center" valign="top">0.5475 (78)</td>
<td align="center" valign="top">0.5000 (80)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>FVS, forward variable selection algorithm; MSE, mean squared error; ROI, region of interest. Entries are sorted in order of ascending MSE values.</p>
</table-wrap-foot>
</table-wrap>
<p>The variable selection model had a very strong effect (<italic>F</italic><sub>2,20</sub>&#x2009;=&#x2009;225.521; <italic>p</italic>&#x2009;&#x003C;&#x2009;0.001; partial <italic>&#x03B7;</italic><sup>2</sup>&#x2009;=&#x2009;0.958). <italic>Post Hoc</italic> analyses revealed that across the 11 models, Boruta improved the decoding accuracy over the &#x2018;without FVS&#x2019; (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001; Cohen&#x2019;s <italic>d</italic>&#x2009;=&#x2009;0.815), as was expected. Beyond the Boruta algorithm, the use of the FVS algorithm significantly improved the performance against both &#x2018;without FVS&#x2019; (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001; Cohen&#x2019;s <italic>d</italic>&#x2009;=&#x2009;2.216) and &#x2018;with Boruta&#x2019; (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001; Cohen&#x2019;s <italic>d</italic>&#x2009;=&#x2009;1.177) with very large effect size (see <xref rid="fig2" ref-type="fig">Figure 2</xref>). Notably, it is essential to consider the computational cost (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S1</xref> for the computational cost of each model and algorithm). In terms of decoding accuracy, &#x2018;random forest (second best)&#x2019; and &#x2018;Gaussian process (third best)&#x2019; models <italic>without</italic> variable selection comparatively were as good as the LassoLars regression model. However, the computational costs of the random forest (CPU time&#x2009;=&#x2009;242.0&#x2009;s without FVS) and Gaussian process (CPU time&#x2009;=&#x2009;80.2&#x2009;s without FVS) models were more excessive because the parameters were more complex than those of LassoLars regression (CPU time&#x2009;=&#x2009;20.3&#x2009;s without FVS). The other models, such as ridge, elastic net, and Lars regression, had faster computations but did not satisfy reasonable performance. Therefore, we focused on the LassoLars regression model for the next step of the analysis.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Performance comparison of 11 regression models with Boruta algorithm, with and without forward variable selection (FVS) to predict age, controlling for total intracranial volume (TIV). Left (blue): 11 regression models without the FVS algorithm. Middle (orange): 11 regression models on a subset of brain regions selected with the Boruta algorithm. Right (green): 11 regression models on a subset of brain regions selected with the FVS algorithm. <italic>P</italic> values were calculated using one-way repeated measures ANOVA tests with Benjamini&#x2013;Hochberg correction for multiple comparisons for 11 pairs of models. &#x002A;<italic>p</italic>&#x2009;&#x003C;&#x2009;0.05, &#x002A;&#x002A;<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01, &#x002A;&#x002A;&#x002A;<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001.</p>
</caption>
<graphic xlink:href="fninf-17-1266713-g002.tif"/>
</fig>
<p>Among all model comparisons, 54 out of 246 brain regions were identified with a Spearman correlation coefficient of 0.63 (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001, <xref rid="fig3" ref-type="fig">Figure 3</xref>) using the FVS-supported LassoLar regression model. The Boruta-supported LassoLar regression model showed a comparable accuracy with a Spearman correlation coefficient of 0.51 (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001). Comparing the FVS and Boruta algorithms, both algorithms commonly selected 20 ROIs, such as the thalamus, hippocampus, amygdala, orbital gyrus, and superior frontal gyrus (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S3</xref>). However, there were several differences in the selected features. While the &#x2018;FVS-supported LassoLar&#x2019; model uniquely selected several ROIs such as parahippocampal gyrus, insula gyrus, basal ganglia, and angular gyrus (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S4</xref>), the &#x2018;Boruta-supported LassoLar&#x2019; model selected inferior parietal gyrus, inferior temporal gyrus, inferior frontal gyrus (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S5</xref>). Focusing only on the three best FVS-supported models (namely, LassoLar, Random Forest, and Gaussian Process), several commonly selected ROIs are thought to be important for age: thalamus, hippocampus, and insula cortex. As was the case for the differences in FVS and Boruta algorithms, different ROIs were selected by each model (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Tables S4, S5</xref> for the details).</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Performance comparison of LassoLar regression with Boruta algorithm, with and without the forward variable selection (FVS) algorithm to predict age, controlling for the effects of total intracranial volume (TIV). Left panel: LassoLar regression with all of brain regions (MSE&#x2009;=&#x2009;0.45, Spearman <italic>&#x03C1;</italic>&#x2009;=&#x2009;0.44, <italic>p</italic>&#x2009;=&#x2009;0.064). Middle panel: LassoLar regression on a subset of brain regions selected with the Boruta algorithm (MSE&#x2009;=&#x2009;0.4, Spearman <italic>&#x03C1;</italic>&#x2009;=&#x2009;0.51, <italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001). Right panel: LassoLar regression on a subset of brain regions selected with the FVS algorithm (MSE&#x2009;=&#x2009;0.36, Spearman <italic>&#x03C1;</italic>&#x2009;=&#x2009;0.63, <italic>p</italic>&#x2009;&#x003C;&#x2009;0.0001). Predicted age data are plotted as a function of the true score. The blue lines and blue shades represent a linear regression line with a confidence interval.</p>
</caption>
<graphic xlink:href="fninf-17-1266713-g003.tif"/>
</fig>
<p><xref rid="fig4" ref-type="fig">Figure 4</xref> shows selected brain regions identified by the FVS-supported LassoLars model. As it turned out, these results were consistent with previous reports on the association of brain regions with age. For example, the thalamus plays a critical role in the coordination of information flow in the brain, mediating communication and integrating many processes, including memory, attention, and perception. Thus, age-related cognitive capability could be associated with micro- and macrostructural alterations in the thalamus. A number of previous studies have shown that increasing age significantly influences the changes in the thalamus (<xref ref-type="bibr" rid="ref28">Good et al., 2001</xref>; <xref ref-type="bibr" rid="ref33">Hutton et al., 2009</xref>).</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Selected brain regions significantly associated with age. The red color denotes a positive correlation with age; the green color denotes a negative correlation with age. <bold>(A)</bold> Premotor thalamus (left), <bold>(B)</bold> premotor thalamus (right), <bold>(C)</bold> sensory thalamus (left), <bold>(D)</bold> orbital gyrus lateral area 11, <bold>(E)</bold> orbital gyrus orbital area 12/47, <bold>(F)</bold> basal ganglia dorsolateral putamen.</p>
</caption>
<graphic xlink:href="fninf-17-1266713-g004.tif"/>
</fig>
</sec>
<sec id="sec10">
<label>3.2.</label>
<title>Improved accuracy for classification models to identify sex</title>
<p><xref rid="tab3" ref-type="table">Table 3</xref> summarizes the accuracies for each ML model with the Boruta algorithm, with and without the FVS algorithm that classifies the male and female groups. The best performance among the comparisons without the FVS algorithm, with the highest accuracy of 75.44%, was obtained using the random forest classifier. Accuracy values were normally distributed for all variable selection algorithms (Lilliefors corrected Shapiro&#x2013;Wilk test all <italic>p</italic>&#x2009;&#x003E;&#x2009;0.31). There was a very strong effect of the variable selection model (<italic>F</italic><sub>2,12</sub>&#x2009;=&#x2009;79.843; <italic>p</italic>&#x2009;&#x003C;&#x2009;0.001; partial <italic>&#x03B7;</italic><sup>2</sup>&#x2009;=&#x2009;0.930). <italic>Post Hoc</italic> analyses revealed that across the 7 models, Boruta improved the decoding accuracy over the &#x2018;without FVS&#x2019; (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001; Cohen&#x2019;s <italic>d</italic>&#x2009;=&#x2009;0.389) as was expected. Beyond the Boruta algorithm, the use of the FVS algorithm significantly improved the performance against both &#x2018;without FVS&#x2019; (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001; Cohen&#x2019;s <italic>d</italic>&#x2009;=&#x2009;1.394) and &#x2018;with Boruta&#x2019; (<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001; Cohen&#x2019;s <italic>d</italic>&#x2009;=&#x2009;0.985) with very large effect size (see <xref rid="fig5" ref-type="fig">Figure 5</xref>).</p>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>Accuracies of the ML models used to classify the male and female groups.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Model</th>
<th align="center" valign="top">Accuracy (%) without FVS (#ROIs)</th>
<th align="center" valign="top">Accuracy (%) with Boruta (#ROIs)</th>
<th align="center" valign="top">Accuracy (%) with FVS (#ROIs)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Random forest</td>
<td align="center" valign="top">75.44 (246)</td>
<td align="center" valign="top">78.35 (73)</td>
<td align="center" valign="top">82.63 (87)</td>
</tr>
<tr>
<td align="left" valign="top">Extreme gradient boosting</td>
<td align="center" valign="top">74.55 (246)</td>
<td align="center" valign="top">77.21 (110)</td>
<td align="center" valign="top">81.13 (92)</td>
</tr>
<tr>
<td align="left" valign="top">Logistic regression with the absolute norm L1</td>
<td align="center" valign="top">70.65 (246)</td>
<td align="center" valign="top">72.46 (88)</td>
<td align="center" valign="top">80.23 (98)</td>
</tr>
<tr>
<td align="left" valign="top">Gradient boosting</td>
<td align="center" valign="top">69.46 (246)</td>
<td align="center" valign="top">71.53 (81)</td>
<td align="center" valign="top">79.04 (76)</td>
</tr>
<tr>
<td align="left" valign="top">Extremely randomized trees</td>
<td align="center" valign="top">68.56 (246)</td>
<td align="center" valign="top">69.62 (102)</td>
<td align="center" valign="top">76.04 (81)</td>
</tr>
<tr>
<td align="left" valign="top">Decision tree</td>
<td align="center" valign="top">66.39 (246)</td>
<td align="center" valign="top">67.18 (77)</td>
<td align="center" valign="top">70.65 (68)</td>
</tr>
<tr>
<td align="left" valign="top">Na&#x00EF;ve bayes</td>
<td align="center" valign="top">61.37 (246)</td>
<td align="center" valign="top">63.75 (64)</td>
<td align="center" valign="top">67.76 (53)</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>FVS, forward variable selection algorithm; ROI, region of interest. Entries are sorted in order of descending accuracy values.</p>
</table-wrap-foot>
</table-wrap>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Performance comparison of 7 classification models with Boruta algorithm, with and without the forward variable selection (FVS) algorithm to classify male and female groups, controlling for the effects of total intracranial volume (TIV). Left (blue): 7 classification models without the FVS algorithm. Middle (orange): 7 classification models on a subset of brain regions selected by the Boruta algorithm. Right (green): 7 classification models on a subset of brain regions selected by the FVS algorithm. <italic>P</italic> values were calculated using one-way repeated measures ANOVA tests with Benjamini&#x2013;Hochberg correction for multiple comparisons. &#x002A;<italic>p</italic>&#x2009;&#x003C;&#x2009;0.05, &#x002A;&#x002A;<italic>p</italic>&#x2009;&#x003C;&#x2009;0.01, &#x002A;&#x002A;&#x002A;<italic>p</italic>&#x2009;&#x003C;&#x2009;0.001.</p>
</caption>
<graphic xlink:href="fninf-17-1266713-g005.tif"/>
</fig>
<p>Notably, it is essential to consider the computational cost (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S2</xref> for the computational cost of each model and algorithm). In terms of decoding accuracy, &#x2018;extreme gradient boosting (second best)&#x2019; models (74.55%) <italic>without</italic> variable selection comparatively were as good as the random forest model (75.44%, see <xref rid="tab3" ref-type="table">Table 3</xref>). However, the computational costs of the extreme gradient boosting model (CPU time&#x2009;=&#x2009;270.0&#x2009;s without FVS) model were more expensive because the parameters were more complex than those of the random forest classifier (CPU time&#x2009;=&#x2009;150.0&#x2009;s without FVS). Additionally, logistic regression with the absolute norm L1 achieved a fairly comparable performance (70.65%) to the extreme gradient boosting classifier (74.55%), while the computational time of logistic regression was considerably shorter (CPU time&#x2009;=&#x2009;19.0&#x2009;s without FVS) than that of extreme gradient boosting classifier (CPU time&#x2009;=&#x2009;270.0&#x2009;s without FVS). Conversely, the extremely randomized trees and na&#x00EF;ve Bayes models had poor performances with low accuracy values (68.56 and 61.37%).</p>
<p>Among all model comparisons, <xref rid="fig6" ref-type="fig">Figure 6</xref> shows that 87 out of 246 brain regions were identified, and the accuracy improved to 82.63% using the FVS-supported random forest classifier. Females were identified with an accuracy of 87%, and males were identified with an accuracy of 77%. The Boruta-supported random forest classifier identified 73 out of 246 brain regions and achieved an accuracy of 78.35%. <xref rid="fig7" ref-type="fig">Figure 7</xref> shows that the selected brain regions, such as the thalamus, inferior frontal gyrus, precuneus, and basal ganglia, were mapped on the Brainnetome Atlas. These brain regions were identified by our model and were consistent with previous reports. For example, a number of studies showed that females had significantly greater volumes in the inferior frontal gyrus, thalamus, and precuneus. Conversely, males had significantly greater volumes in the basal ganglia and lingual gyrus. Comparing the FVS and Boruta algorithms, both algorithms commonly selected 36 ROIs, such as the thalamus, inferior frontal gyrus, inferior parietal gyrus, basal ganglia, and middle frontal gyrus. However, there were several differences in the selected features (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S6</xref>). While the &#x2018;FVS-supported Random Forest&#x2019; model uniquely selected several ROIs such as superior frontal gyrus, superior parietal gyrus, fusiform gyrus, and cingulate (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S7</xref>), the &#x2018;Boruta-supported Random Forest&#x2019; model selected orbital gyrus, postcentral gyrus, lateral occipital gyrus (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Table S8</xref>). Focusing only on the two best FVS-supported models (namely, Random Forest and extreme gradient boosting), several commonly selected ROIs are thought to be important for sex: thalamus, cingulate, and inferior frontal gyrus. As was the case for the differences in FVS and Boruta algorithms, different ROIs were selected by each model (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Tables S7, S8</xref> for details).</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Performance comparison of the random forest classifier with Boruta algorithm, with and without the forward variable selection (FVS) algorithm to classify two groups, controlling for the effects of total intracranial volume (TIV). Left panel: random forest classifier analysis with all of brain regions. Middle panel: random forest classifier on a subset of brain regions selected by the Boruta algorithm. Right panel: random forest classifier on a subset of brain regions selected by the FVS algorithm.</p>
</caption>
<graphic xlink:href="fninf-17-1266713-g006.tif"/>
</fig>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Selected brain regions identified as predictors of the sex categories (male and female). The red color denotes male predicting volume&#x2009;&#x003E;&#x2009;female predicting volume; the green color denotes male predicting volume&#x2009;&#x003C;&#x2009;female predicting volume. <bold>(A)</bold> Premotor thalamus, <bold>(B)</bold> inferior frontal gyrus dorsal area 44, <bold>(C)</bold> precuneus medial area 5 (PEm), <bold>(D)</bold> basal ganglia dorsolateral putamen, <bold>(E)</bold> basal ganglia nucleus accumbens, <bold>(F)</bold> ligual gyrus medio ventral occipital caudal.</p>
</caption>
<graphic xlink:href="fninf-17-1266713-g007.tif"/>
</fig>
</sec>
</sec>
<sec sec-type="discussions" id="sec11">
<label>4.</label>
<title>Discussion</title>
<p>In this study, a parallelized FVS toolbox is developed to provide optimized decoding of neuroimaging data samples. Our toolbox can be used to propose the best ML model for user&#x2019;s input data and identify a small group of important features that significantly improve the performance of the ML model. We have demonstrated that the toolbox is feasible for region of interest (ROI) data without revising the model types (parametric or nonparametric) and parameter settings, suggesting that this toolbox is generalizable and could potentially be used to train multiple types of neuroimaging data without modification. Given previous use cases of the ML model that have been established in genetic studies using the FVS algorithm (<xref ref-type="bibr" rid="ref14">Dang and Kishino, 2022</xref>), we have extended the FVS algorithm and the toolbox has been created for neuroimaging studies. To examine the feasibility of our ML pipelines, sample neuroimaging data were acquired from the HCP database. As case samples, we compared the accuracies (predictability) of the classical ML models with and without the FVS algorithm.</p>
<p>We tested the performances of several ML models by analyzing large structural MRI datasets with a large number of variables (246 brain regions). An easy-to-use computational package may help novel data scientists in neuroimaging research and advance the research by identifying accurate features relevant to questions of interest.</p>
<sec id="sec12">
<label>4.1.</label>
<title>Comparison against existing methods</title>
<p>The proposed method presents the following advantages compared with the previous methods. First, neuroscientists could avoid decision uncertainties when considering or choosing the most appropriate model for their own datasets. In our proposed method, users only provide the input data and decide whether to run the proposed ML pipeline for either classification or regression based on their purpose of analysis. The automatic algorithm will rank the ML models and recommend the best model for the user&#x2019;s dataset. In this study, the results showed that random forest was the most accurate model for classification. Random forest is an ML model that is based on combining multiple decision trees by random selection of samples. Therefore, random forest overcomes the problem of overfitting decision trees, which can result in a better fitting of the model (<xref ref-type="bibr" rid="ref27">Ghose et al., 2012</xref>; <xref ref-type="bibr" rid="ref48">Mitra et al., 2014</xref>; <xref ref-type="bibr" rid="ref61">Sarica et al., 2017</xref>; <xref ref-type="bibr" rid="ref76">Zhu et al., 2018</xref>). In the regression task, the best performance for predicting the age of healthy individuals was obtained using the LassoLar model. The performance of random forest was ranked second (<xref ref-type="bibr" rid="ref64">Smith et al., 2013</xref>; <xref ref-type="bibr" rid="ref37">Jog et al., 2017</xref>; <xref ref-type="bibr" rid="ref17">Dimitriadis et al., 2018</xref>). In the second step, the FVS algorithm was used to select a feature (e.g., ROI) that improves the accuracies of ML classification models or reduces the MSEs of ML regression models at each iteration. This procedure was stopped if the performance of the ML model reached a maximization. The FVS algorithm attempts to identify a minimal core set of brain regions that can provide insights into brain functions. The results showed that the performances of all ML models in classification and regression were significantly improved after applying the FVS algorithm. For example, the FVS algorithms identified 87 ROI features that improved the accuracy of the random forest classifier from 75.44 to 82.63%.</p>
</sec>
<sec id="sec13">
<label>4.2.</label>
<title>Advantages of FVS</title>
<p>In the regression model, the option with FVS significantly outperformed the option without FVS and with the Boruta algorithm (<xref rid="fig2" ref-type="fig">Figures 2</xref>, <xref rid="fig5" ref-type="fig">5</xref>, respectively). For the regression model to identify age, the LassoLars model was selected as the best model, and 54 regions to account for age were identified. In brief, this finding suggests that the thalamus and orbital gyrus are significantly associated with age-related changes. A previous study found that a general linear model identified age-related changes in terms of gray matter density (<xref ref-type="bibr" rid="ref65">Tisserand et al., 2004</xref>). The prefrontal cortex (PFC), the (medial) temporal lobe, and the posterior parietal cortex showed the greatest differences in gray matter density.</p>
<p>For the classification model to identify regions that account for sex, the random forest model was determined to be the best model. This result suggests that the inferior frontal gyrus, thalamus, and precuneus regions may contribute to identifying sex. A previous study (<xref ref-type="bibr" rid="ref73">Xu et al., 2000</xref>) suggested that the posterior right frontal lobe, right temporal lobe, left basal ganglia, parietal lobe, and cerebellum regions may contribute to identifying differences between males and females.</p>
</sec>
<sec id="sec14">
<label>4.3.</label>
<title>Limitations</title>
<p>Although it was apparent that the FVS algorithm robustly and significantly improved the accuracy for both classification and regression models, the downside of this model is the computation time to apply nearly all possible pairs to consider all features (up to the specified number of pairs specified by the user). To compensate for the issue of time, the parallel computing pipelines implemented in our toolbox effectively minimize and compensate for the computational time.</p>
<p>While the FVS algorithm significantly improves the performances of the ML models, the computational burden of the FVS algorithm is still a difficult challenge for personal computers. Even if we apply the parallel computational techniques to overcome large-scale problems in the FVS algorithm, a high-performance computer, but not on a low-spec computer, is necessary to efficiently run our proposed tool. Although the computational speed could be improved, based on the material efficiency aspects of personal computers, implementing our strategy would still not be possible. In the future, we may implement a new method (<xref ref-type="bibr" rid="ref72">Xing et al., 2016</xref>) that can balance high-speed computation and material efficiency.</p>
<p>Furthermore, instead of a voxel-based approach, atlas-based analyses were performed in this study for demonstrational purposes. One could apply the proposed method to voxel-based datasets in future studies. It has been shown that the differential outcomes between voxel-based and atlas-based analyses to identify structural brain alterations between groups (<xref ref-type="bibr" rid="ref63">Seyedi et al., 2020</xref>). Therefore, the reported observations in this study may differ from those using a voxel-based approach. The best model for determining age in our study, for example, identified several brain regions reported to be associated with age (i.e., thalamus, hippocampus, amygdala, orbital gyrus, and superior frontal gyrus) as reported in the previous works (<xref ref-type="bibr" rid="ref28">Good et al., 2001</xref>; <xref ref-type="bibr" rid="ref75">Zhou et al., 2022</xref>). However, certain brain regions reported in these studies were not chosen via our approach; these include the postcentral gyrus, superior temporal gyrus, brainstem, medial frontal cortex, middle temporal gyrus, middle frontal gyrus, and cerebellum. The same holds true for the classification models for sex. Although the amygdala, precuneus, cerebellum, parietal operculum cortex, and orbital cortex were reported as significant regions to classify sex, we only found a limited overlap such as the thalamus, inferior frontal gyrus, inferior parietal gyrus, basal ganglia (<xref ref-type="bibr" rid="ref58">Ruigrok et al., 2014</xref>).</p>
<p>Although direct comparisons of decoding accuracies were made, it would be important to be aware that the oFVSD, or ML in general, may not necessarily identify the same brain regions as previous studies. While our data-driven feature selection approach certainly benefits from blind-folded neural decoding, on the other hand, the FVS approach is rather greedy, and it may lead to local minimas and may not necessarily reflect scientific rigorousness based on existing evidence. Our toolbox may further practically and logically benefit from human supervision based on existing literature by restricting target features to scientifically validated brain regions of interest (<xref ref-type="bibr" rid="ref13">Chu et al., 2012</xref>). That said, users of the oFVSD need to carefully interpret the outcome due to the pitfalls of the data-driven ML approach that this toolbox may offer.</p>
</sec>
<sec id="sec15">
<label>4.4.</label>
<title>Computational time</title>
<p>The high-dimensional problems of these datasets will result in more difficult challenges for the FVS algorithms. The FVS algorithm was applied to analyze the 16S rRNA sequencing microbiome datasets, where the number of features was huge (approximately 30,000 features) in our previous study (<xref ref-type="bibr" rid="ref14">Dang and Kishino, 2022</xref>). To reduce the computational burden of the FVS algorithm, some prescreening algorithms [such as the Boruta algorithm (<xref ref-type="bibr" rid="ref40">Kursa and Rudnicki, 2010</xref>) and Laplacian score (<xref ref-type="bibr" rid="ref31">He et al., 2005</xref>)] were proposed to detect all strongly and weakly relevant features to reduce the considerable data dimensionality. With the initial prescreening pipeline, the computational time of the FVS algorithm could be significantly decreased from days to hours (<xref ref-type="bibr" rid="ref14">Dang and Kishino, 2022</xref>). The current pipeline uses a fixed prescreening model. Therefore, additional considerations of this strategy may be necessary if the number of features becomes very large to apply a rigorous search method such as our approach.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec16">
<label>5.</label>
<title>Conclusion</title>
<p>The use of neuroimaging data to train ML models has a significant potential for identifying brain regions whose structure and activities may contain information predictive of physical phenotypes, mental states, and pathological conditions. However, an overwhelmingly large number of ML models exist, which may increase the difficulties for those unfamiliar with mathematical theories. Moreover, the high dimensionality of neuroimaging data negatively impacts the power of ML models to discover hidden information in the selected neural resources. Furthermore, researchers are often challenged with time-consuming computations to identify neural substrates, a variety of neuroscientific discoveries, and the development of novel therapeutic interventions. In this study, we proposed a novel procedure that not only automatically selects the best ML model for specific neuroimaging data but also identifies a group of brain regions that substantially improve the performance in terms of high-speed computation and high accuracy. This powerful decoding tool may be applicable to a variety of neuroimaging modalities.</p>
</sec>
<sec sec-type="data-availability" id="sec17">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref rid="SM1" ref-type="supplementary-material">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="sec18">
<title>Author contributions</title>
<p>TD: Conceptualization, Formal analysis, Investigation, Methodology, Software, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. AF: Formal analysis, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. MM: Conceptualization, Data curation, Funding acquisition, Project administration, Supervision, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec sec-type="funding-information" id="sec19">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by JSPS KAKENHI (Grant Number JP21J21850), the JST COI Grant Numbers: (JPMJCE1311 and JPMJCA2208), and the Moonshot R&#x0026;D Goal 9 (JPMJMS2296).</p>
</sec>
<sec sec-type="COI-statement" id="sec20">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="sec100" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack>
<p>We thank Haruka Kobayashi for checking source codes in Github.</p>
</ack>
<sec sec-type="supplementary-material" id="sec21">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fninf.2023.1266713/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fninf.2023.1266713/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<fn-group>
<fn id="fn0001">
<p><sup>1</sup><ext-link xlink:href="https://atlas.brainnetome.org/bnatlas.html" ext-link-type="uri">https://atlas.brainnetome.org/bnatlas.html</ext-link>
</p>
</fn>
<fn id="fn0002">
<p><sup>2</sup><ext-link xlink:href="https://github.com/tungtokyo1108/FVS_decoder" ext-link-type="uri">https://github.com/tungtokyo1108/FVS_decoder</ext-link>
</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Agrawal</surname> <given-names>T.</given-names></name></person-group> (<year>2021</year>). &#x201C;<article-title>Hyperparameter optimization using Scikit-learn</article-title>&#x201D; in <source>Hyperparameter optimization in machine learning</source> (<publisher-loc>Berkeley, CA</publisher-loc>: <publisher-name>Apress</publisher-name>), <fpage>31</fpage>&#x2013;<lpage>51</lpage>.</citation></ref>
<ref id="ref2"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-Nesf</surname> <given-names>M. A. Y.</given-names></name> <name><surname>Abdesselem</surname> <given-names>H. B.</given-names></name> <name><surname>Bensmail</surname> <given-names>I.</given-names></name> <name><surname>Ibrahim</surname> <given-names>S.</given-names></name> <name><surname>Saeed</surname> <given-names>W. A. H.</given-names></name> <name><surname>Mohammed</surname> <given-names>S. S. I.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Prognostic tools and candidate drugs based on plasma proteomics of patients with severe COVID-19 complications</article-title>. <source>Nat. Commun.</source> <volume>13</volume>:<fpage>946</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-022-28639-4</pub-id>, PMID: <pub-id pub-id-type="pmid">35177642</pub-id></citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bergstra</surname> <given-names>J.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name></person-group> (<year>2012</year>). <article-title>Random search for hyper-parameter optimization</article-title>. <source>J. Mach. Learn. Res.</source> <volume>13</volume>, <fpage>281</fpage>&#x2013;<lpage>305</lpage>. doi: <pub-id pub-id-type="doi">10.5555/2188385.2188395</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Bisong</surname> <given-names>E.</given-names></name></person-group> (<year>2019</year>). &#x201C;<article-title>More supervised machine learning techniques with Scikit-learn</article-title>&#x201D; in <source>Building machine learning and deep learning models on Google cloud platform</source> (<publisher-loc>Berkeley, United States</publisher-loc>: <publisher-name>Apress</publisher-name>), <fpage>287</fpage>&#x2013;<lpage>308</lpage>.</citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Blanco</surname> <given-names>R.</given-names></name> <name><surname>Larra&#x00F1;aga</surname> <given-names>P.</given-names></name> <name><surname>Inza</surname> <given-names>I.</given-names></name> <name><surname>Sierra</surname> <given-names>B.</given-names></name></person-group> (<year>2004</year>). <article-title>Gene selection for cancer classification using wrapper approaches</article-title>. <source>Int. J. Pattern Recognit. Artif. Intell.</source> <volume>18</volume>, <fpage>1373</fpage>&#x2013;<lpage>1390</lpage>. doi: <pub-id pub-id-type="doi">10.1142/S0218001404003800</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boulesteix</surname> <given-names>A.-L.</given-names></name> <name><surname>Janitza</surname> <given-names>S.</given-names></name> <name><surname>Kruppa</surname> <given-names>J.</given-names></name> <name><surname>K&#x00F6;nig</surname> <given-names>I. R.</given-names></name></person-group> (<year>2012</year>). <article-title>Overview of random forest methodology and practical guidance with emphasis on computational biology and bioinformatics: random forests in bioinformatics</article-title>. <source>Wiley Interdiscip. Rev. Data Min. Knowl. Discov.</source> <volume>2</volume>, <fpage>493</fpage>&#x2013;<lpage>507</lpage>. doi: <pub-id pub-id-type="doi">10.1002/widm.1072</pub-id></citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Breiman</surname> <given-names>L.</given-names></name></person-group> (<year>2001</year>). <article-title>Random forests</article-title>. <source>Mach. Learn.</source> <volume>45</volume>, <fpage>5</fpage>&#x2013;<lpage>32</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1010933404324</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="book"><person-group person-group-type="author"><name><surname>B&#x00FC;hlmann</surname> <given-names>P.</given-names></name> <name><surname>van de Geer</surname> <given-names>S. A.</given-names></name></person-group> (<year>2011</year>). <source>Statistics for high-dimensional data: Methods, theory and applications, springer series in statistics</source>. <publisher-name>Springer</publisher-name>: <publisher-loc>Heidelberg; New York</publisher-loc>.</citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Burnham</surname> <given-names>K. P.</given-names></name> <name><surname>Anderson</surname> <given-names>D. R.</given-names></name></person-group> (<year>2004</year>). <article-title>Multimodel inference: understanding AIC and BIC in model selection</article-title>. <source>Sociol. Methods Res.</source> <volume>33</volume>, <fpage>261</fpage>&#x2013;<lpage>304</lpage>. doi: <pub-id pub-id-type="doi">10.1177/0049124104268644</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chandrashekar</surname> <given-names>G.</given-names></name> <name><surname>Sahin</surname> <given-names>F.</given-names></name></person-group> (<year>2014</year>). <article-title>A survey on feature selection methods</article-title>. <source>Comput. Electr. Eng.</source> <volume>40</volume>, <fpage>16</fpage>&#x2013;<lpage>28</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compeleceng.2013.11.024</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>T.</given-names></name> <name><surname>Guestrin</surname> <given-names>C.</given-names></name></person-group> (<year>2016</year>). <article-title>XGBoost: a scalable tree boosting system, in: proceedings of the 22nd ACM SIGKDD international conference on knowledge discovery and data mining</article-title>. <conf-name>Presented at the KDD&#x2019;16: The 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining</conf-name>. <publisher-loc>San Francisco California USA</publisher-loc>: <publisher-name>ACM</publisher-name>. pp. <fpage>785</fpage>&#x2013;<lpage>794</lpage>.</citation></ref>
<ref id="ref12"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chu</surname> <given-names>W.</given-names></name> <name><surname>Ghahramani</surname> <given-names>Z.</given-names></name> <name><surname>Falciani</surname> <given-names>F.</given-names></name> <name><surname>Wild</surname> <given-names>D. L.</given-names></name></person-group> (<year>2005</year>). <article-title>Biomarker discovery in microarray gene expression data with Gaussian processes</article-title>. <source>Bioinformatics</source> <volume>21</volume>, <fpage>3385</fpage>&#x2013;<lpage>3393</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/bti526</pub-id>, PMID: <pub-id pub-id-type="pmid">15937031</pub-id></citation></ref>
<ref id="ref13"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chu</surname> <given-names>C.</given-names></name> <name><surname>Hsu</surname> <given-names>A.-L.</given-names></name> <name><surname>Chou</surname> <given-names>K.-H.</given-names></name> <name><surname>Bandettini</surname> <given-names>P.</given-names></name> <name><surname>Lin</surname> <given-names>C.</given-names></name></person-group> (<year>2012</year>). <article-title>Does feature selection improve classification accuracy? Impact of sample size and feature selection on classification using anatomical magnetic resonance images</article-title>. <source>Neuroimage</source> <volume>60</volume>, <fpage>59</fpage>&#x2013;<lpage>70</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2011.11.066</pub-id>, PMID: <pub-id pub-id-type="pmid">22166797</pub-id></citation></ref>
<ref id="ref14"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dang</surname> <given-names>T.</given-names></name> <name><surname>Kishino</surname> <given-names>H.</given-names></name></person-group> (<year>2022</year>). <article-title>Forward variable selection improves the power of random Forest for high-dimensional Micro biome data</article-title>. <source>J. Cancer Sci. Clin. Ther.</source> <volume>6</volume>, <fpage>87</fpage>&#x2013;<lpage>105</lpage>. doi: <pub-id pub-id-type="doi">10.26502/jcsct.5079147</pub-id></citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dang</surname> <given-names>T.</given-names></name> <name><surname>Kumaishi</surname> <given-names>K.</given-names></name> <name><surname>Usui</surname> <given-names>E.</given-names></name> <name><surname>Kobori</surname> <given-names>S.</given-names></name> <name><surname>Sato</surname> <given-names>T.</given-names></name> <name><surname>Toda</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Stochastic variational variable selection for high-dimensional microbiome data</article-title>. <source>Microbiome</source> <volume>10</volume>:<fpage>236</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s40168-022-01439-0</pub-id>, PMID: <pub-id pub-id-type="pmid">36566203</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Diamond</surname> <given-names>S.</given-names></name> <name><surname>Andeer</surname> <given-names>P. F.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Crits-Christoph</surname> <given-names>A.</given-names></name> <name><surname>Burstein</surname> <given-names>D.</given-names></name> <name><surname>Anantharaman</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Mediterranean grassland soil C-N compound turnover is dependent on rainfall and depth, and is mediated by genomically divergent microorganisms</article-title>. <source>Nat. Microbiol.</source> <volume>4</volume>, <fpage>1356</fpage>&#x2013;<lpage>1367</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41564-019-0449-y</pub-id>, PMID: <pub-id pub-id-type="pmid">31110364</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dimitriadis</surname> <given-names>S. I.</given-names></name> <name><surname>Liparas</surname> <given-names>D.</given-names></name> <name><surname>Tsolaki</surname> <given-names>M. N.</given-names></name></person-group> (<year>2018</year>). <article-title>Random forest feature selection, fusion and ensemble strategy: combining multiple morphological MRI measures to discriminate among healhy elderly, MCI, cMCI and alzheimer&#x2019;s disease patients: from the alzheimer&#x2019;s disease neuroimaging initiative (ADNI) database</article-title>. <source>J. Neurosci. Methods</source> <volume>302</volume>, <fpage>14</fpage>&#x2013;<lpage>23</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jneumeth.2017.12.010</pub-id>, PMID: <pub-id pub-id-type="pmid">29269320</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Edwinson</surname> <given-names>A. L.</given-names></name> <name><surname>Yang</surname> <given-names>L.</given-names></name> <name><surname>Peters</surname> <given-names>S.</given-names></name> <name><surname>Hanning</surname> <given-names>N.</given-names></name> <name><surname>Jeraldo</surname> <given-names>P.</given-names></name> <name><surname>Jagtap</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Gut microbial &#x03B2;-glucuronidases regulate host luminal proteases and are depleted in irritable bowel syndrome</article-title>. <source>Nat. Microbiol.</source> <volume>7</volume>, <fpage>680</fpage>&#x2013;<lpage>694</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41564-022-01103-1</pub-id>, PMID: <pub-id pub-id-type="pmid">35484230</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Efron</surname> <given-names>B.</given-names></name> <name><surname>Hastie</surname> <given-names>T.</given-names></name> <name><surname>Johnstone</surname> <given-names>I.</given-names></name> <name><surname>Tibshirani</surname> <given-names>R.</given-names></name></person-group> (<year>2004</year>). <article-title>Least angle regression</article-title>. <source>Ann. Stat.</source> <volume>32</volume>, <fpage>407</fpage>&#x2013;<lpage>499</lpage>. doi: <pub-id pub-id-type="doi">10.1214/009053604000000067</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eshaghi</surname> <given-names>A.</given-names></name> <name><surname>Wottschel</surname> <given-names>V.</given-names></name> <name><surname>Cortese</surname> <given-names>R.</given-names></name> <name><surname>Calabrese</surname> <given-names>M.</given-names></name> <name><surname>Sahraian</surname> <given-names>M. A.</given-names></name> <name><surname>Thompson</surname> <given-names>A. J.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Gray matter MRI differentiates neuromyelitis optica from multiple sclerosis using random forest</article-title>. <source>Neurology</source> <volume>87</volume>, <fpage>2463</fpage>&#x2013;<lpage>2470</lpage>. doi: <pub-id pub-id-type="doi">10.1212/WNL.0000000000003395</pub-id>, PMID: <pub-id pub-id-type="pmid">27807185</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fan</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Zhuo</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>The human Brainnetome atlas: a new brain atlas based on connectional architecture</article-title>. <source>Cereb. Cortex</source> <volume>26</volume>, <fpage>3508</fpage>&#x2013;<lpage>3526</lpage>. doi: <pub-id pub-id-type="doi">10.1093/cercor/bhw157</pub-id>, PMID: <pub-id pub-id-type="pmid">27230218</pub-id></citation></ref>
<ref id="ref22"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Ferreira</surname> <given-names>A. J.</given-names></name> <name><surname>Figueiredo</surname> <given-names>M. A. T.</given-names></name></person-group> (<year>2012</year>). &#x201C;<article-title>Ensemble machine learning</article-title>&#x201D; in <source>Methods and applications</source>. eds. <person-group person-group-type="editor"><name><surname>Zhang</surname> <given-names>C.</given-names></name> <name><surname>Ma</surname> <given-names>Y.</given-names></name></person-group> (<publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name>)</citation></ref>
<ref id="ref23"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Filli</surname> <given-names>L.</given-names></name> <name><surname>Rosskopf</surname> <given-names>A. B.</given-names></name> <name><surname>Sutter</surname> <given-names>R.</given-names></name> <name><surname>Fucentese</surname> <given-names>S. F.</given-names></name> <name><surname>Pfirrmann</surname> <given-names>C. W.</given-names></name></person-group> (<year>2018</year>). <article-title>MRI predictors of posterolateral corner instability: a decision tree analysis of patients with acute anterior cruciate ligament tear</article-title>. <source>Radiology</source> <volume>289</volume>, <fpage>170</fpage>&#x2013;<lpage>180</lpage>. doi: <pub-id pub-id-type="doi">10.1148/radiol.2018180194</pub-id></citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Friedman</surname> <given-names>J. H.</given-names></name></person-group> (<year>2001</year>). <article-title>Greedy function approximation: a gradient boosting machine</article-title>. <source>Ann. Stat.</source> <volume>29</volume>, <fpage>1189</fpage>&#x2013;<lpage>1232</lpage>. doi: <pub-id pub-id-type="doi">10.1214/aos/1013203451</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>Gavankar</surname> <given-names>S. S.</given-names></name> <name><surname>Sawarkar</surname> <given-names>S. D.</given-names></name></person-group> (<year>2017</year>). <article-title>Eager decision tree</article-title>. <conf-name>Proceedings of the 2017 2nd International Conference for Convergence in Technology (I2CT)</conf-name>. <publisher-name>IEEE</publisher-name>: <publisher-loc>Mumbai</publisher-loc>. pp. <fpage>837</fpage>&#x2013;<lpage>840</lpage>.</citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Geurts</surname> <given-names>P.</given-names></name> <name><surname>Ernst</surname> <given-names>D.</given-names></name> <name><surname>Wehenkel</surname> <given-names>L.</given-names></name></person-group> (<year>2006</year>). <article-title>Extremely randomized trees</article-title>. <source>Mach. Learn.</source> <volume>63</volume>, <fpage>3</fpage>&#x2013;<lpage>42</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10994-006-6226-1</pub-id></citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ghose</surname> <given-names>S.</given-names></name> <name><surname>Mitra</surname> <given-names>J.</given-names></name> <name><surname>Oliver</surname> <given-names>A.</given-names></name> <name><surname>Marti</surname> <given-names>R.</given-names></name> <name><surname>Llad&#x00F3;</surname> <given-names>X.</given-names></name> <name><surname>Freixenet</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>A random forest based classification approach to prostate segmentation in MRI</article-title>. <source>MICCAI Grand Chall. Prostate MR Image Segmentation</source> <volume>2012</volume>, <fpage>125</fpage>&#x2013;<lpage>128</lpage>.</citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Good</surname> <given-names>C. D.</given-names></name> <name><surname>Johnsrude</surname> <given-names>I. S.</given-names></name> <name><surname>Ashburner</surname> <given-names>J.</given-names></name> <name><surname>Henson</surname> <given-names>R. N.</given-names></name> <name><surname>Friston</surname> <given-names>K. J.</given-names></name> <name><surname>Frackowiak</surname> <given-names>R. S.</given-names></name></person-group> (<year>2001</year>). <article-title>A voxel-based morphometric study of ageing in 465 normal adult human brains</article-title>. <source>Neuroimage</source> <volume>14</volume>, <fpage>21</fpage>&#x2013;<lpage>36</lpage>. doi: <pub-id pub-id-type="doi">10.1006/nimg.2001.0786</pub-id>, PMID: <pub-id pub-id-type="pmid">11525331</pub-id></citation></ref>
<ref id="ref29"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guyon</surname> <given-names>I.</given-names></name> <name><surname>Elisseeff</surname> <given-names>A.</given-names></name></person-group> (<year>2003</year>). <article-title>An introduction to variable and feature selection</article-title>. <source>J. Mach. Learn. Res.</source> <volume>3</volume>, <fpage>1157</fpage>&#x2013;<lpage>1182</lpage>. doi: <pub-id pub-id-type="doi">10.1162/153244303322753616</pub-id></citation></ref>
<ref id="ref30"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Hastie</surname> <given-names>T.</given-names></name> <name><surname>Tibshirani</surname> <given-names>R.</given-names></name> <name><surname>Friedman</surname> <given-names>J. H.</given-names></name></person-group> (<year>2009</year>). <source>The elements of statistical learning: Data mining, inference, and prediction</source>, <edition>2nd ed.</edition>. Springer series in statistics. <publisher-loc>New York, NY</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation></ref>
<ref id="ref31"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>He</surname> <given-names>X.</given-names></name> <name><surname>Cai</surname> <given-names>D.</given-names></name> <name><surname>Niyogi</surname> <given-names>P.</given-names></name></person-group> (<year>2005</year>). <article-title>Laplacian score for feature selection</article-title>. <conf-name>Proceedings of the International Conference on Neural Information Processing Systems</conf-name>. <publisher-loc>Cambridge</publisher-loc>: <publisher-name>MIT Press</publisher-name>.</citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>J.</given-names></name> <name><surname>Breheny</surname> <given-names>P.</given-names></name> <name><surname>Ma</surname> <given-names>S.</given-names></name></person-group> (<year>2012</year>). <article-title>A selective review of group selection in high-dimensional models</article-title>. <source>Stat. Sci.</source> <volume>27</volume>, <fpage>481</fpage>&#x2013;<lpage>499</lpage>. doi: <pub-id pub-id-type="doi">10.1214/12-STS392</pub-id>, PMID: <pub-id pub-id-type="pmid">24174707</pub-id></citation></ref>
<ref id="ref33"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hutton</surname> <given-names>C.</given-names></name> <name><surname>Draganski</surname> <given-names>B.</given-names></name> <name><surname>Ashburner</surname> <given-names>J.</given-names></name> <name><surname>Weiskopf</surname> <given-names>N.</given-names></name></person-group> (<year>2009</year>). <article-title>A comparison between voxel-based cortical thickness and voxel-based morphometry in normal aging</article-title>. <source>Neuroimage</source> <volume>48</volume>, <fpage>371</fpage>&#x2013;<lpage>380</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2009.06.043</pub-id>, PMID: <pub-id pub-id-type="pmid">19559801</pub-id></citation></ref>
<ref id="ref34"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jain</surname> <given-names>A. K.</given-names></name> <name><surname>Duin</surname> <given-names>P. W.</given-names></name> <name><surname>Mao</surname> <given-names>J.</given-names></name></person-group> (<year>2000</year>). <article-title>Statistical pattern recognition: a review</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>22</volume>, <fpage>4</fpage>&#x2013;<lpage>37</lpage>. doi: <pub-id pub-id-type="doi">10.1109/34.824819</pub-id></citation></ref>
<ref id="ref35"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Janssen</surname> <given-names>R. J.</given-names></name> <name><surname>Mour&#x00E3;o-Miranda</surname> <given-names>J.</given-names></name> <name><surname>Schnack</surname> <given-names>H. G.</given-names></name></person-group> (<year>2018</year>). <article-title>Making individual prognoses in psychiatry using neuroimaging and machine learning</article-title>. <source>Biol. Psychiatry. Cogn. Neurosci. Neuroimaging</source> <volume>3</volume>, <fpage>798</fpage>&#x2013;<lpage>808</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bpsc.2018.04.004</pub-id>, PMID: <pub-id pub-id-type="pmid">29789268</pub-id></citation></ref>
<ref id="ref36"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jirapech-Umpai</surname> <given-names>T.</given-names></name> <name><surname>Aitken</surname> <given-names>S.</given-names></name></person-group> (<year>2005</year>). <article-title>Feature selection and classification for microarray data analysis: evolutionary methods for identifying predictive genes</article-title>. <source>BMC Bioinformatics</source> <volume>6</volume>, <fpage>148</fpage>&#x2013;<lpage>111</lpage>. doi: <pub-id pub-id-type="doi">10.1186/1471-2105-6-148</pub-id></citation></ref>
<ref id="ref37"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jog</surname> <given-names>A.</given-names></name> <name><surname>Carass</surname> <given-names>A.</given-names></name> <name><surname>Roy</surname> <given-names>S.</given-names></name> <name><surname>Pham</surname> <given-names>D. L.</given-names></name> <name><surname>Prince</surname> <given-names>J. L.</given-names></name></person-group> (<year>2017</year>). <article-title>Random forest regression for magnetic resonance image synthesis</article-title>. <source>Med. Image Anal.</source> <volume>35</volume>, <fpage>475</fpage>&#x2013;<lpage>488</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.media.2016.08.009</pub-id>, PMID: <pub-id pub-id-type="pmid">27607469</pub-id></citation></ref>
<ref id="ref38"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>Y. H.</given-names></name> <name><surname>Kim</surname> <given-names>M.-J.</given-names></name> <name><surname>Shin</surname> <given-names>H. J.</given-names></name> <name><surname>Yoon</surname> <given-names>H.</given-names></name> <name><surname>Han</surname> <given-names>S. J.</given-names></name> <name><surname>Koh</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>MRI-based decision tree model for diagnosis of biliary atresia</article-title>. <source>Eur. Radiol.</source> <volume>28</volume>, <fpage>3422</fpage>&#x2013;<lpage>3431</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00330-018-5327-0</pub-id>, PMID: <pub-id pub-id-type="pmid">29476221</pub-id></citation></ref>
<ref id="ref39"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kuncheva</surname> <given-names>L. I.</given-names></name> <name><surname>Rodriguez</surname> <given-names>J. J.</given-names></name> <name><surname>Plumpton</surname> <given-names>C. O.</given-names></name> <name><surname>Linden</surname> <given-names>D. E. J.</given-names></name> <name><surname>Johnston</surname> <given-names>S. J.</given-names></name></person-group> (<year>2010</year>). <article-title>Random subspace ensembles for FMRI classification</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>29</volume>, <fpage>531</fpage>&#x2013;<lpage>542</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2009.2037756</pub-id>, PMID: <pub-id pub-id-type="pmid">20129853</pub-id></citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kursa</surname> <given-names>M. B.</given-names></name> <name><surname>Rudnicki</surname> <given-names>W. R.</given-names></name></person-group> (<year>2010</year>). <article-title>Feature selection with the Boruta package</article-title>. <source>J. Stat. Softw.</source> <volume>36</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.18637/jss.v036.i11</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Kutner</surname> <given-names>M. H.</given-names></name> <name><surname>Nachtsheim</surname> <given-names>C.</given-names></name> <name><surname>Neter</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name></person-group>, (<year>2005</year>). <source>Applied linear statistical models</source>, <edition>5th ed</edition> <publisher-name>McGraw-Hill Irwin</publisher-name>, <publisher-loc>Boston</publisher-loc>.</citation></ref>
<ref id="ref42"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liao</surname> <given-names>J. G.</given-names></name> <name><surname>Chin</surname> <given-names>K.-V.</given-names></name></person-group> (<year>2007</year>). <article-title>Logistic regression for disease classification using microarray data: model selection in a large p and small n case</article-title>. <source>Bioinformatics</source> <volume>23</volume>, <fpage>1945</fpage>&#x2013;<lpage>1951</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btm287</pub-id>, PMID: <pub-id pub-id-type="pmid">17540680</pub-id></citation></ref>
<ref id="ref43"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mayneris-Perxachs</surname> <given-names>J.</given-names></name> <name><surname>Castells-Nobau</surname> <given-names>A.</given-names></name> <name><surname>Arnoriaga-Rodr&#x00ED;guez</surname> <given-names>M.</given-names></name> <name><surname>Martin</surname> <given-names>M.</given-names></name> <name><surname>de la Vega-Correa</surname> <given-names>L.</given-names></name> <name><surname>Zapata</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Microbiota alterations in proline metabolism impact depression</article-title>. <source>Cell Metab.</source> <volume>34</volume>, <fpage>681</fpage>&#x2013;<lpage>701.e10</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cmet.2022.04.001</pub-id>, PMID: <pub-id pub-id-type="pmid">35508109</pub-id></citation></ref>
<ref id="ref44"><citation citation-type="confproc"><person-group person-group-type="author"><name><surname>McCallum</surname> <given-names>A.</given-names></name> <name><surname>Nigam</surname> <given-names>K.</given-names></name></person-group>, (<year>1998</year>). <article-title>A comparison of event models for naive bayes text classification</article-title>. <conf-name>Proceedings in Workshop on Learning for Text Categorization, AAAI&#x2019;98</conf-name>, <conf-loc>Madison, WI</conf-loc>. pp. <fpage>41</fpage>&#x2013;<lpage>48</lpage>.</citation></ref>
<ref id="ref45"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McIntosh</surname> <given-names>A. R.</given-names></name> <name><surname>Bookstein</surname> <given-names>F. L.</given-names></name> <name><surname>Haxby</surname> <given-names>J. V.</given-names></name> <name><surname>Grady</surname> <given-names>C. L.</given-names></name></person-group> (<year>1996</year>). <article-title>Spatial pattern analysis of functional brain images using partial least squares</article-title>. <source>Neuroimage</source> <volume>3</volume>, <fpage>143</fpage>&#x2013;<lpage>157</lpage>. doi: <pub-id pub-id-type="doi">10.1006/nimg.1996.0016</pub-id></citation></ref>
<ref id="ref46"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>McIntosh</surname> <given-names>A. R.</given-names></name> <name><surname>Lobaugh</surname> <given-names>N. J.</given-names></name></person-group> (<year>2004</year>). <article-title>Partial least squares analysis of neuroimaging data: applications and advances</article-title>. <source>Neuroimage</source> <volume>23</volume>, <fpage>S250</fpage>&#x2013;<lpage>S263</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2004.07.020</pub-id>, PMID: <pub-id pub-id-type="pmid">15501095</pub-id></citation></ref>
<ref id="ref47"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Metwaly</surname> <given-names>A.</given-names></name> <name><surname>Dunkel</surname> <given-names>A.</given-names></name> <name><surname>Waldschmitt</surname> <given-names>N.</given-names></name> <name><surname>Raj</surname> <given-names>A. C. D.</given-names></name> <name><surname>Lagkouvardos</surname> <given-names>I.</given-names></name> <name><surname>Corraliza</surname> <given-names>A. M.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Integrated microbiota and metabolite profiles link Crohn&#x2019;s disease to sulfur metabolism</article-title>. <source>Nat. Commun.</source> <volume>11</volume>:<fpage>4322</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-020-17956-1</pub-id>, PMID: <pub-id pub-id-type="pmid">32859898</pub-id></citation></ref>
<ref id="ref48"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mitra</surname> <given-names>J.</given-names></name> <name><surname>Bourgeat</surname> <given-names>P.</given-names></name> <name><surname>Fripp</surname> <given-names>J.</given-names></name> <name><surname>Ghose</surname> <given-names>S.</given-names></name> <name><surname>Rose</surname> <given-names>S.</given-names></name> <name><surname>Salvado</surname> <given-names>O.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Lesion segmentation from multimodal MRI using random forest following ischemic stroke</article-title>. <source>Neuroimage</source> <volume>98</volume>, <fpage>324</fpage>&#x2013;<lpage>335</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2014.04.056</pub-id>, PMID: <pub-id pub-id-type="pmid">24793830</pub-id></citation></ref>
<ref id="ref49"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mwangi</surname> <given-names>B.</given-names></name> <name><surname>Tian</surname> <given-names>T. S.</given-names></name> <name><surname>Soares</surname> <given-names>J. C.</given-names></name></person-group> (<year>2014</year>). <article-title>A review of feature reduction techniques in neuroimaging</article-title>. <source>Neuroinformatics</source> <volume>12</volume>, <fpage>229</fpage>&#x2013;<lpage>244</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s12021-013-9204-3</pub-id>, PMID: <pub-id pub-id-type="pmid">24013948</pub-id></citation></ref>
<ref id="ref50"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Naik</surname> <given-names>J.</given-names></name> <name><surname>Patel</surname> <given-names>S.</given-names></name></person-group> (<year>2014</year>). <article-title>Tumor detection and classification using decision tree in brain MRI</article-title>. <source>Int. J. Comput. Sci. Netw. Secur. Ijcsns</source> <volume>14</volume>:<fpage>87</fpage>.</citation></ref>
<ref id="ref51"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nielsen</surname> <given-names>A. N.</given-names></name> <name><surname>Barch</surname> <given-names>D. M.</given-names></name> <name><surname>Petersen</surname> <given-names>S. E.</given-names></name> <name><surname>Schlaggar</surname> <given-names>B. L.</given-names></name> <name><surname>Greene</surname> <given-names>D. J.</given-names></name></person-group> (<year>2020</year>). <article-title>Machine learning with neuroimaging: evaluating its applications in psychiatry</article-title>. <source>Biol. Psychiatry Cogn. Neurosci. Neuroimaging</source> <volume>5</volume>, <fpage>791</fpage>&#x2013;<lpage>798</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bpsc.2019.11.007</pub-id>, PMID: <pub-id pub-id-type="pmid">31982357</pub-id></citation></ref>
<ref id="ref52"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>O&#x2019;Toole</surname> <given-names>A. J.</given-names></name> <name><surname>Jiang</surname> <given-names>F.</given-names></name> <name><surname>Abdi</surname> <given-names>H.</given-names></name> <name><surname>P&#x00E9;nard</surname> <given-names>N.</given-names></name> <name><surname>Dunlop</surname> <given-names>J. P.</given-names></name> <name><surname>Parent</surname> <given-names>M. A.</given-names></name></person-group> (<year>2007</year>). <article-title>Theoretical, statistical, and practical perspectives on pattern-based classification approaches to the analysis of functional neuroimaging data</article-title>. <source>J. Cogn. Neurosci.</source> <volume>19</volume>, <fpage>1735</fpage>&#x2013;<lpage>1752</lpage>. doi: <pub-id pub-id-type="doi">10.1162/jocn.2007.19.11.1735</pub-id>, PMID: <pub-id pub-id-type="pmid">17958478</pub-id></citation></ref>
<ref id="ref53"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ooi</surname> <given-names>C. H.</given-names></name> <name><surname>Tan</surname> <given-names>P.</given-names></name></person-group> (<year>2003</year>). <article-title>Genetic algorithms applied to multi-class prediction for the analysis of gene expression data</article-title>. <source>Bioinformatics</source> <volume>19</volume>, <fpage>37</fpage>&#x2013;<lpage>44</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/19.1.37</pub-id>, PMID: <pub-id pub-id-type="pmid">12499291</pub-id></citation></ref>
<ref id="ref54"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Palach</surname> <given-names>J.</given-names></name></person-group> (<year>2014</year>). <source>Parallel programming with Python: develop efficient parallel systems using the robust Python environment, Community experience distilled</source>. <publisher-name>Packt Publishing</publisher-name>: <publisher-loc>Birmingham</publisher-loc>.</citation></ref>
<ref id="ref55"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pereira</surname> <given-names>F.</given-names></name> <name><surname>Mitchell</surname> <given-names>T.</given-names></name> <name><surname>Botvinick</surname> <given-names>M.</given-names></name></person-group> (<year>2009</year>). <article-title>Machine learning classifiers and fMRI: a tutorial overview</article-title>. <source>NeuroImage</source> <volume>45</volume>, <fpage>S199</fpage>&#x2013;<lpage>S209</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2008.11.007</pub-id>, PMID: <pub-id pub-id-type="pmid">19070668</pub-id></citation></ref>
<ref id="ref56"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pietzner</surname> <given-names>M.</given-names></name> <name><surname>Wheeler</surname> <given-names>E.</given-names></name> <name><surname>Carrasco-Zanini</surname> <given-names>J.</given-names></name> <name><surname>Kerrison</surname> <given-names>N. D.</given-names></name> <name><surname>Oerton</surname> <given-names>E.</given-names></name> <name><surname>Koprulu</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Synergistic insights into human health from aptamer- and antibody-based proteomic profiling</article-title>. <source>Nat. Commun.</source> <volume>12</volume>:<fpage>6822</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-021-27164-0</pub-id>, PMID: <pub-id pub-id-type="pmid">34819519</pub-id></citation></ref>
<ref id="ref57"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Rasmussen</surname> <given-names>C. E.</given-names></name></person-group> (<year>2003</year>). &#x201C;<article-title>Gaussian processes in machine learning</article-title>&#x201D; in <source>Summer school on machine learning</source>. eds. <person-group person-group-type="editor"><name><surname>Bousquet</surname> <given-names>O.</given-names></name> <name><surname>von Luxburg</surname> <given-names>U.</given-names></name> <name><surname>R&#x00E4;tsch</surname> <given-names>G.</given-names></name></person-group> (<publisher-loc>Berlin, Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>63</fpage>&#x2013;<lpage>71</lpage>.</citation></ref>
<ref id="ref58"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ruigrok</surname> <given-names>A. N. V.</given-names></name> <name><surname>Salimi-Khorshidi</surname> <given-names>G.</given-names></name> <name><surname>Lai</surname> <given-names>M.-C.</given-names></name> <name><surname>Baron-Cohen</surname> <given-names>S.</given-names></name> <name><surname>Lombardo</surname> <given-names>M. V.</given-names></name> <name><surname>Tait</surname> <given-names>R. J.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>A meta-analysis of sex differences in human brain structure</article-title>. <source>Neurosci. Biobehav. Rev.</source> <volume>39</volume>, <fpage>34</fpage>&#x2013;<lpage>50</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neubiorev.2013.12.004</pub-id>, PMID: <pub-id pub-id-type="pmid">24374381</pub-id></citation></ref>
<ref id="ref59"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saeys</surname> <given-names>Y.</given-names></name> <name><surname>Inza</surname> <given-names>I.</given-names></name> <name><surname>Larranaga</surname> <given-names>P.</given-names></name></person-group> (<year>2007</year>). <article-title>A review of feature selection techniques in bioinformatics</article-title>. <source>Bioinformatics</source> <volume>23</volume>, <fpage>2507</fpage>&#x2013;<lpage>2517</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btm344</pub-id></citation></ref>
<ref id="ref60"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saffouri</surname> <given-names>G. B.</given-names></name> <name><surname>Shields-Cutler</surname> <given-names>R. R.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>Y.</given-names></name> <name><surname>Lekatz</surname> <given-names>H. R.</given-names></name> <name><surname>Hale</surname> <given-names>V. L.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Small intestinal microbial dysbiosis underlies symptoms associated with functional gastrointestinal disorders</article-title>. <source>Nat. Commun.</source> <volume>10</volume>:<fpage>2012</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-019-09964-7</pub-id>, PMID: <pub-id pub-id-type="pmid">31043597</pub-id></citation></ref>
<ref id="ref61"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sarica</surname> <given-names>A.</given-names></name> <name><surname>Cerasa</surname> <given-names>A.</given-names></name> <name><surname>Quattrone</surname> <given-names>A.</given-names></name></person-group> (<year>2017</year>). <article-title>Random Forest algorithm for the classification of neuroimaging data in Alzheimer&#x2019;s disease: a systematic review</article-title>. <source>Front. Aging Neurosci.</source> <volume>9</volume>:<fpage>329</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnagi.2017.00329</pub-id>, PMID: <pub-id pub-id-type="pmid">29056906</pub-id></citation></ref>
<ref id="ref62"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Scott</surname> <given-names>D. W.</given-names></name></person-group>, (<year>1992</year>). <source>Multivariate density estimation: theory, practice, and visualization</source>. <publisher-name>Wiley</publisher-name>: <publisher-loc>New York</publisher-loc></citation></ref>
<ref id="ref63"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Seyedi</surname> <given-names>S.</given-names></name> <name><surname>Jafari</surname> <given-names>R.</given-names></name> <name><surname>Talaei</surname> <given-names>A.</given-names></name> <name><surname>Naseri</surname> <given-names>S.</given-names></name> <name><surname>Momennezhad</surname> <given-names>M.</given-names></name> <name><surname>Moghaddam</surname> <given-names>M. D.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Comparing VBM and ROI analyses for detection of gray matter abnormalities in patients with bipolar disorder using MRI</article-title>. <source>Middle East Curr. Psychiatry</source> <volume>27</volume>:<fpage>69</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s43045-020-00076-3</pub-id></citation></ref>
<ref id="ref64"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>P. F.</given-names></name> <name><surname>Ganesh</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>P.</given-names></name></person-group> (<year>2013</year>). <article-title>A comparison of random forest regression and multiple linear regression for prediction in neuroscience</article-title>. <source>J. Neurosci. Methods</source> <volume>220</volume>, <fpage>85</fpage>&#x2013;<lpage>91</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jneumeth.2013.08.024</pub-id>, PMID: <pub-id pub-id-type="pmid">24012917</pub-id></citation></ref>
<ref id="ref65"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tisserand</surname> <given-names>D. J.</given-names></name> <name><surname>van Boxtel</surname> <given-names>M. P. J.</given-names></name> <name><surname>Pruessner</surname> <given-names>J. C.</given-names></name> <name><surname>Hofman</surname> <given-names>P.</given-names></name> <name><surname>Evans</surname> <given-names>A. C.</given-names></name> <name><surname>Jolles</surname> <given-names>J.</given-names></name></person-group> (<year>2004</year>). <article-title>A voxel-based morphometric study to determine individual differences in gray matter density associated with age and cognitive change over time</article-title>. <source>Cereb. Cortex</source> <volume>14</volume>, <fpage>966</fpage>&#x2013;<lpage>973</lpage>. doi: <pub-id pub-id-type="doi">10.1093/cercor/bhh057</pub-id>, PMID: <pub-id pub-id-type="pmid">15115735</pub-id></citation></ref>
<ref id="ref66"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Vovk</surname> <given-names>V</given-names></name></person-group>. (<year>2013</year>). <source>Empirical inference</source>. <publisher-name>Springer</publisher-name>: <publisher-loc>Berlin Heidelberg, New York, NY</publisher-loc>.</citation></ref>
<ref id="ref67"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vul</surname> <given-names>E.</given-names></name> <name><surname>Harris</surname> <given-names>C.</given-names></name> <name><surname>Winkielman</surname> <given-names>P.</given-names></name> <name><surname>Pashler</surname> <given-names>H.</given-names></name></person-group> (<year>2009</year>). <article-title>Puzzlingly high correlations in fMRI studies of emotion, personality, and social cognition</article-title>. <source>Perspect. Psychol. Sci.</source> <volume>4</volume>, <fpage>274</fpage>&#x2013;<lpage>290</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1745-6924.2009.01125.x</pub-id>, PMID: <pub-id pub-id-type="pmid">26158964</pub-id></citation></ref>
<ref id="ref68"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Warren</surname> <given-names>S. L.</given-names></name> <name><surname>Moustafa</surname> <given-names>A. A.</given-names></name></person-group> (<year>2022</year>). <article-title>Functional magnetic resonance imaging, deep learning, and Alzheimer&#x2019;s disease: a systematic review</article-title>. <source>J. Neuroimaging Off. J. Am. Soc. Neuroimaging</source> <volume>33</volume>, <fpage>5</fpage>&#x2013;<lpage>18</lpage>. doi: <pub-id pub-id-type="doi">10.1111/jon.13063</pub-id>, PMID: <pub-id pub-id-type="pmid">36257926</pub-id></citation></ref>
<ref id="ref69"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wassermann</surname> <given-names>D.</given-names></name> <name><surname>Bloy</surname> <given-names>L.</given-names></name> <name><surname>Kanterakis</surname> <given-names>E.</given-names></name> <name><surname>Verma</surname> <given-names>R.</given-names></name> <name><surname>Deriche</surname> <given-names>R.</given-names></name></person-group> (<year>2010</year>). <article-title>Unsupervised white matter fiber clustering and tract probability map generation: applications of a Gaussian process framework for white matter fibers</article-title>. <source>Neuroimage</source> <volume>51</volume>, <fpage>228</fpage>&#x2013;<lpage>241</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2010.01.004</pub-id>, PMID: <pub-id pub-id-type="pmid">20079439</pub-id></citation></ref>
<ref id="ref70"><citation citation-type="book"><person-group person-group-type="author"><name><surname>Weisberg</surname> <given-names>S.</given-names></name></person-group>, (<year>2005</year>). <source>Applied linear regression: weisberg/applied linear regression 3e, Wiley series in probability and statistics</source>. <publisher-name>John Wiley &#x0026; Sons, Inc</publisher-name>: <publisher-loc>Hoboken, NJ, USA</publisher-loc></citation></ref>
<ref id="ref71"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>T. T.</given-names></name> <name><surname>Chen</surname> <given-names>Y. F.</given-names></name> <name><surname>Hastie</surname> <given-names>T.</given-names></name> <name><surname>Sobel</surname> <given-names>E.</given-names></name> <name><surname>Lange</surname> <given-names>K.</given-names></name></person-group> (<year>2009</year>). <article-title>Genome-wide association analysis by lasso penalized logistic regression</article-title>. <source>Bioinformatics</source> <volume>25</volume>, <fpage>714</fpage>&#x2013;<lpage>721</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btp041</pub-id>, PMID: <pub-id pub-id-type="pmid">19176549</pub-id></citation></ref>
<ref id="ref72"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xing</surname> <given-names>E. P.</given-names></name> <name><surname>Ho</surname> <given-names>Q.</given-names></name> <name><surname>Xie</surname> <given-names>P.</given-names></name> <name><surname>Wei</surname> <given-names>D.</given-names></name></person-group> (<year>2016</year>). <article-title>Strategies and principles of distributed machine learning on big data</article-title>. <source>Engineering</source> <volume>2</volume>, <fpage>179</fpage>&#x2013;<lpage>195</lpage>. doi: <pub-id pub-id-type="doi">10.1016/J.ENG.2016.02.008</pub-id></citation></ref>
<ref id="ref73"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Kobayashi</surname> <given-names>S.</given-names></name> <name><surname>Yamaguchi</surname> <given-names>S.</given-names></name> <name><surname>Iijima</surname> <given-names>K.</given-names></name> <name><surname>Okada</surname> <given-names>K.</given-names></name> <name><surname>Yamashita</surname> <given-names>K.</given-names></name></person-group> (<year>2000</year>). <article-title>Gender effects on age-related changes in brain structure</article-title>. <source>AJNR Am. J. Neuroradiol.</source> <volume>21</volume>, <fpage>112</fpage>&#x2013;<lpage>118</lpage>. PMID: <pub-id pub-id-type="pmid">10669234</pub-id></citation></ref>
<ref id="ref74"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yousef</surname> <given-names>M.</given-names></name> <name><surname>Jung</surname> <given-names>S.</given-names></name> <name><surname>Kossenkov</surname> <given-names>A. V.</given-names></name> <name><surname>Showe</surname> <given-names>L. C.</given-names></name> <name><surname>Showe</surname> <given-names>M. K.</given-names></name></person-group> (<year>2007</year>). <article-title>Na&#x00EF;ve Bayes for micro RNA target predictions&#x2014;machine learning for microRNA targets</article-title>. <source>Bioinformatics</source> <volume>23</volume>, <fpage>2987</fpage>&#x2013;<lpage>2992</lpage>. doi: <pub-id pub-id-type="doi">10.1093/bioinformatics/btm484</pub-id>, PMID: <pub-id pub-id-type="pmid">17925304</pub-id></citation></ref>
<ref id="ref75"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>X.</given-names></name> <name><surname>Wu</surname> <given-names>R.</given-names></name> <name><surname>Zeng</surname> <given-names>Y.</given-names></name> <name><surname>Qi</surname> <given-names>Z.</given-names></name> <name><surname>Ferraro</surname> <given-names>S.</given-names></name> <name><surname>Xu</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Choice of voxel-based morphometry processing pipeline drives variability in the location of neuroanatomical brain markers</article-title>. <source>Commun. Biol.</source> <volume>5</volume>:<fpage>913</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s42003-022-03880-1</pub-id>, PMID: <pub-id pub-id-type="pmid">36068295</pub-id></citation></ref>
<ref id="ref76"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>X.</given-names></name> <name><surname>Du</surname> <given-names>X.</given-names></name> <name><surname>Kerich</surname> <given-names>M.</given-names></name> <name><surname>Lohoff</surname> <given-names>F. W.</given-names></name> <name><surname>Momenan</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>Random forest based classification of alcohol dependence patients and healthy controls using resting state MRI</article-title>. <source>Neurosci. Lett.</source> <volume>676</volume>, <fpage>27</fpage>&#x2013;<lpage>33</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neulet.2018.04.007</pub-id>, PMID: <pub-id pub-id-type="pmid">29626649</pub-id></citation></ref>
<ref id="ref77"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>G.</given-names></name> <name><surname>Jiang</surname> <given-names>B.</given-names></name> <name><surname>Tong</surname> <given-names>L.</given-names></name> <name><surname>Xie</surname> <given-names>Y.</given-names></name> <name><surname>Zaharchuk</surname> <given-names>G.</given-names></name> <name><surname>Wintermark</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>Applications of deep learning to neuro-imaging techniques</article-title>. <source>Front. Neurol.</source> <volume>10</volume>:<fpage>869</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fneur.2019.00869</pub-id>, PMID: <pub-id pub-id-type="pmid">31474928</pub-id></citation></ref>
<ref id="ref78"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>H.</given-names></name> <name><surname>Hastie</surname> <given-names>T.</given-names></name></person-group> (<year>2005</year>). <article-title>Regularization and variable selection via the elastic net</article-title>. <source>J. R. Stat. Soc. Ser. B</source> <volume>67</volume>, <fpage>301</fpage>&#x2013;<lpage>320</lpage>. doi: <pub-id pub-id-type="doi">10.1111/j.1467-9868.2005.00503.x</pub-id></citation></ref>
</ref-list>
</back>
</article>