<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2017.00460</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Diagnosing Autism Spectrum Disorder from Brain Resting-State Functional Connectivity Patterns Using a Deep Neural Network with a Novel Feature Selection Method</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Guo</surname> <given-names>Xinyu</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/299445/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Dominick</surname> <given-names>Kelli C.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/425336/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Minai</surname> <given-names>Ali A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/41710/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Hailong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Erickson</surname> <given-names>Craig A.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Lu</surname> <given-names>Long J.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/350736/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Division of Biomedical Informatics, Cincinnati Children&#x00027;s Hospital Research Foundation</institution> <country>Cincinnati, OH, United States</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Electrical Engineering and Computing Systems, University of Cincinnati</institution> <country>Cincinnati, OH, United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>The Kelly O&#x00027;Leary Center for Autism Spectrum Disorders, Cincinnati Children&#x00027;s Hospital Medical Center</institution> <country>Cincinnati, OH, United States</country></aff>
<aff id="aff4"><sup>4</sup><institution>School of Information Management, Wuhan University</institution> <country>Wuhan, China</country></aff>
<aff id="aff5"><sup>5</sup><institution>Department of Environmental Health, College of Medicine, University of Cincinnati</institution> <country>Cincinnati, OH, United States</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Habib Benali, Concordia University, Canada</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Jie Shi, MathWorks, United States; Fabienne Samson, Universit&#x000E9; de Montr&#x000E9;al, Canada</p></fn>
<fn fn-type="corresp" id="fn001"><p>&#x0002A;Correspondence: Long J. Lu <email>long.lu&#x00040;cchmc.org</email></p></fn>
<fn fn-type="other" id="fn002"><p>This article was submitted to Brain Imaging Methods, a section of the journal Frontiers in Neuroscience</p></fn></author-notes>
<pub-date pub-type="epub">
<day>21</day>
<month>08</month>
<year>2017</year>
</pub-date>
<pub-date pub-type="collection">
<year>2017</year>
</pub-date>
<volume>11</volume>
<elocation-id>460</elocation-id>
<history>
<date date-type="received">
<day>16</day>
<month>03</month>
<year>2017</year>
</date>
<date date-type="accepted">
<day>31</day>
<month>07</month>
<year>2017</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2017 Guo, Dominick, Minai, Li, Erickson and Lu.</copyright-statement>
<copyright-year>2017</copyright-year>
<copyright-holder>Guo, Dominick, Minai, Li, Erickson and Lu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) or licensor are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract><p>The whole-brain functional connectivity (FC) pattern obtained from resting-state functional magnetic resonance imaging data are commonly applied to study neuropsychiatric conditions such as autism spectrum disorder (ASD) by using different machine learning models. Recent studies indicate that both hyper- and hypo- aberrant ASD-associated FCs were widely distributed throughout the entire brain rather than only in some specific brain regions. Deep neural networks (DNN) with multiple hidden layers have shown the ability to systematically extract lower-to-higher level information from high dimensional data across a series of neural hidden layers, significantly improving classification accuracy for such data. In this study, a DNN with a novel feature selection method (DNN-FS) is developed for the high dimensional whole-brain resting-state FC pattern classification of ASD patients vs. typical development (TD) controls. The feature selection method is able to help the DNN generate low dimensional high-quality representations of the whole-brain FC patterns by selecting features with high discriminating power from multiple trained sparse auto-encoders. For the comparison, a DNN without the feature selection method (DNN-woFS) is developed, and both of them are tested with different architectures (i.e., with different numbers of hidden layers/nodes). Results show that the best classification accuracy of <bold>86.36%</bold> is generated by the DNN-FS approach with 3 hidden layers and 150 hidden nodes (3/150). Remarkably, DNN-FS outperforms DNN-woFS for all architectures studied. The most significant accuracy improvement was <bold>9.09%</bold> with the 3/150 architecture. The method also outperforms other feature selection methods, e.g., two sample <italic>t</italic>-test and elastic net. In addition to improving the classification accuracy, a Fisher&#x00027;s score-based biomarker identification method based on the DNN is also developed, and used to identify 32 FCs related to ASD. These FCs come from or cross different pre-defined brain networks including the default-mode, cingulo-opercular, frontal-parietal, and cerebellum. Thirteen of them are statically significant between ASD and TD groups (two sample <italic>t</italic>-test <italic>p</italic> &#x0003C; 0.05) while 19 of them are not. The relationship between the statically significant FCs and the corresponding ASD behavior symptoms is discussed based on the literature and clinician&#x00027;s expert knowledge. Meanwhile, the potential reason of obtaining 19 FCs which are not statistically significant is also provided.</p></abstract>
<kwd-group>
<kwd>autism spectrum disorder</kwd>
<kwd>resting-state fMRI</kwd>
<kwd>deep neural network</kwd>
<kwd>sparse auto-encoder</kwd>
<kwd>feature selection</kwd>
</kwd-group>
<contract-num rid="cn001">8UL1TR000077-05</contract-num>
<contract-num rid="cn002">HL111829</contract-num>
<contract-num rid="cn003">31601083</contract-num>
<contract-sponsor id="cn001">National Center for Advancing Translational Sciences<named-content content-type="fundref-id">10.13039/100006108</named-content></contract-sponsor>
<contract-sponsor id="cn002">National Institutes of Health<named-content content-type="fundref-id">10.13039/100000002</named-content></contract-sponsor>
<contract-sponsor id="cn003">National Natural Science Foundation of China<named-content content-type="fundref-id">10.13039/501100001809</named-content></contract-sponsor>
<counts>
<fig-count count="10"/>
<table-count count="6"/>
<equation-count count="19"/>
<ref-count count="75"/>
<page-count count="19"/>
<word-count count="14580"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Autism spectrum disorder (ASD) is a serious lifelong condition characterized by repetitive, restricted behavior as well as deficits in communication and reciprocal social interactions (American Psychiatric Association, <xref ref-type="bibr" rid="B2">2013</xref>). The traditional procedure for diagnosing ASD is largely based on narrative interactions between individuals and clinical professionals (Yahata et al., <xref ref-type="bibr" rid="B71">2016</xref>). Such methods, lacking biological evidence, not only are prone to generate a high variance during the diagnosis (Mandell et al., <xref ref-type="bibr" rid="B47">2007</xref>) but also require a long period to detect abnormalities (Nylander et al., <xref ref-type="bibr" rid="B54">2013</xref>). As a complement to the current behavior-based diagnoses, functional magnetic resonance imaging (fMRI) has been widely applied to explore the functional characteristics or properties of a brain. It is able to assist neuroscientists to get valuable insights into different neurological disorders (Martin et al., <xref ref-type="bibr" rid="B48">2016</xref>). A human brain can be understood as a complex system with various regions performing different functions. Although, some structural regions may not be connected locally, they are integrated globally to process different types of information. fMRI measuring the changes of blood oxygen level-dependent (BOLD) signal in a non-invasive way has been applied to reveal regional associations or brain networks. In 1995, Biswal et al. (<xref ref-type="bibr" rid="B5">1995</xref>) discovered that various brain regions still actively interact with each other while a subject was at rest (not in any cognitive task). Since then, resting-state fMRI (rs-fMRI) has become an important tool for investigating brain networks of different brain disorders such as Alzheimer&#x00027;s disease (Chase, <xref ref-type="bibr" rid="B9">2014</xref>), schizophrenia (Lynall et al., <xref ref-type="bibr" rid="B46">2010</xref>), and ASD (Monk et al., <xref ref-type="bibr" rid="B50">2009</xref>) and has generated many invaluable insights into neural substrates that underlie brain disorders.</p>
<p>Previous ASD studies based on rs-fMRI images have examined anatomical and functional abnormalities associated with ASD from cohorts at different age levels, e.g., for the adolescence cohort (10&#x02013;19 years approximately; Assaf et al., <xref ref-type="bibr" rid="B3">2010</xref>; Keown et al., <xref ref-type="bibr" rid="B37">2013</xref>; Starck et al., <xref ref-type="bibr" rid="B60">2013</xref>; Bos et al., <xref ref-type="bibr" rid="B6">2014</xref>; Chen S. et al., <xref ref-type="bibr" rid="B11">2015</xref>; Doyle-Thomas et al., <xref ref-type="bibr" rid="B17">2015</xref>; Iidaka, <xref ref-type="bibr" rid="B31">2015</xref>; Jann et al., <xref ref-type="bibr" rid="B35">2015</xref>), for the adult cohort (&#x02265;20 years) (Cherkassky et al., <xref ref-type="bibr" rid="B12">2006</xref>; Monk et al., <xref ref-type="bibr" rid="B50">2009</xref>; Tyszka et al., <xref ref-type="bibr" rid="B64">2013</xref>; Itahashi et al., <xref ref-type="bibr" rid="B33">2014</xref>, <xref ref-type="bibr" rid="B34">2015</xref>; Chen C. P. et al., <xref ref-type="bibr" rid="B10">2015</xref>; Jung et al., <xref ref-type="bibr" rid="B36">2015</xref>), and for the cohort covered all age levels (Alaerts et al., <xref ref-type="bibr" rid="B1">2015</xref>; Cerliani et al., <xref ref-type="bibr" rid="B8">2015</xref>). The results helped clarify the relevant neurological foundations of ASD in different age levels. Extant literature suggests that the basic organization of functional networks is similar across all age levels. However, the levels of connectivity and modulation appear altered in ASD. The results suggested that both hypo- and hyper-connectivity occurred in ASD relative to typical developed (TD) controls. By studying an adult cohort, Monk et al. (<xref ref-type="bibr" rid="B50">2009</xref>) found that poorer social functioning in the ASD group was correlated with hypo-connectivity between the posterior cingulate cortex and the superior frontal gyrus, and more severe restricted and repetitive behaviors in ASD were correlated with stronger connectivity between the posterior cingulate cortex and right parahippocampal gyrus. These findings indicated that ASD adults showed altered intrinsic connectivity within the default network, and connectivity between these structures is associated with specific ASD symptoms. Assaf et al. (<xref ref-type="bibr" rid="B3">2010</xref>) discovered that compared to adolescent controls, adolescent ASD patients showed decreased functional connectivities (FCs) between the precuneus and medial prefrontal cortex/anterior cingulate cortex, DMN core areas, and other default mode sub-network areas. The magnitude of FCs in these regions inversely correlated with the severity of patients&#x00027; social and communication deficits. Keown et al. (<xref ref-type="bibr" rid="B37">2013</xref>) found that local FCs were atypically increased in adolescents with ASD in temporal-occipital regions bilaterally by applying rs-fMRI and a graph model. Hyper-connectivity in the posterior brain regions was found to be associated with higher ASD symptom severity. Supekar et al. (<xref ref-type="bibr" rid="B63">2013</xref>) claimed that hyper-connectivity of short range connections in ASD was observed at the whole-brain and subsystems levels. It demonstrated that at earlier ages, the brains of children with ASD are largely functionally hyper-connected in ways that contribute to social dysfunction. Above finding are spread in different brain networks including cingulo-opercular (CO), default-mode (DM), cerebellum (CB), and frontal-parietal (FP).</p>
<p>The above findings indicated that aberrant ASD-associated FCs were widely distributed throughout the entire brain, as opposed to showing a restricted pattern within only a few specific brain regions. Thus, in this research, we developed the machine learning model to explore ASD-related FCs from the whole-brain FC pattern (FCP) which is a set of FCs including each FC between every pair of pre-defined brain regions. The method was tested on the dataset from an adolescent cohort.</p>
<p>Machine learning algorithms have been successfully employed in the automated classification of altered FC patterns related to ASD based on rs-fMRI images (Uddin et al., <xref ref-type="bibr" rid="B67">2013</xref>). For example, Yahata et al. (<xref ref-type="bibr" rid="B71">2016</xref>) developed a classifier achieves high accuracy (85%) for a Japanese discovery cohort and demonstrates a remarkable degree of generalization (75% accuracy) for two independent validation cohorts in the USA and Japan. Models developed by Deshpande et al. (<xref ref-type="bibr" rid="B15">2013</xref>) and Nielsen et al. (<xref ref-type="bibr" rid="B53">2013</xref>) achieved 95.9 and 60% accuracy on a self-collected dataset and the autism brain imaging data exchange I (ABIDE I) dataset, respectively.</p>
<p>As an important machine learning tool, deep learning has been widely applied in various research areas (Krizhevsky et al., <xref ref-type="bibr" rid="B39">2012</xref>; Graves et al., <xref ref-type="bibr" rid="B23">2013</xref>). Stacking several auto-encoders (AEs) or restricted Boltzmann machine (RBMs) is a common way of building a DNN. The DNN architecture is able to improve the feature learning capacity by exploring latent or hidden low-dimensional representations which are inherent in high-dimensional raw data. With such representations, the classification model performance can be enhanced effectively. The DNN is widely used in neuroimaging studies (Brosch et al., <xref ref-type="bibr" rid="B7">2013</xref>; Suk et al., <xref ref-type="bibr" rid="B61">2014</xref>; Kim et al., <xref ref-type="bibr" rid="B38">2016</xref>). Hjelm et al. (<xref ref-type="bibr" rid="B29">2014</xref>) demonstrated that the DNN constructed by stacking RBMs extracted better spatial and temporal information from fMRI data compared with ICA and PCA algorithms. Suk et al. (<xref ref-type="bibr" rid="B62">2015</xref>) stacked several AEs to build the DNN which was applied in a task of discriminating AD patients from mild cognitive impairment patients, and obtained the accuracy of 83.7% on the ADNI dataset. Plis et al. (<xref ref-type="bibr" rid="B55">2013</xref>) conducted a validation study on structural and functional neuroimaging data to show that a stacked RBMs can learn physiologically important representations and detect latent relations in neuroimaging data. Although, the DNN obtained many competitive results, its feature learning capacity is still able to be improved. Recently, researchers have tried using multiple AEs to learn better representations of the data. Zhu et al. (<xref ref-type="bibr" rid="B74">2014</xref>) developed a DNN-based framework to learn multi-channel deep feature representations for the face recognition task. In their design, multiple deep learning networks are applied to extract representations of different parts of the face, e.g., eyes, nose and mouth, and the combined representations used to train a classification model. Similar studies can be found in literature (Hong et al., <xref ref-type="bibr" rid="B30">2015</xref>; Zhao et al., <xref ref-type="bibr" rid="B73">2015</xref>). Inspired by ideas in above papers, a novel feature selection method based on multiple SAEs was proposed. Different from above studies, we constructed one SAE with high feature learning capacity by selecting features from multiple trained SAEs instead of just stacking multiple SAEs together to form several DNNs.</p>
<p>In this study, a DNN-based classification model has been developed for distinguishing ASD patients from TD controls based on the whole-brain FCP of each subject. The DNN model consists of the stacked SAEs for data dimension reduction, and a softmax regression (SR) on the top of the stacked SAEs for the classification. Moreover, a novel feature selection method based on multiple trained sparse auto-encoders (SAEs) has been developed. It can initialize weights between the first hidden layer and the input layer (the first feature layer) of the DNN by selecting a set of dissimilar features with high discriminative power from multiple trained SAEs. In this study, the first objective is to enhance the performance of the DNN for classifying ASD patients and TD controls by applying a novel feature selection method. The DNN-FS (3/150 hidden layers/nodes) obtained the best classification accuracy of <bold>86.36%</bold>. Most importantly, it outperformed the DNN-woFS for each comparison scenario. The most significant improvement, of <bold>9.09%</bold>, occurred when the architecture had three hidden layers with 150 nodes each. Meanwhile, by comparing the average discriminating power of the same layer in both DNN-FS and DNN-woFS, we found that the discriminating ability of any layer in DNN-FS is higher than the discriminating ability of the corresponding layer in DNN-woFS. In addition, using two other feature selection methods&#x02014;a two-sample <italic>t</italic>-test and an elastic net&#x02014;results showed that the proposed method (86.82%) outperformed both the other methods (two-sample <italic>t</italic>-test: 70.67%, elastic net: 79.54%). The second objective is to identify abnormal FCs related to ASD. To achieve the goal, a DNN-based biomarker identification method was developed to locate 32 FCs associated with ASD. Generally, the findings supported the points of view in literature that (1) both hyper- and hypo-connectivity exist in certain brain networks (e.g., DM) of the ASD patients; (2) abnormal between-network connectivity exists in ASD patients. More explanation is given in the Discussion section. For readers&#x00027; convenience, the abbreviations of terminologies used in the following sections are listed in Table <xref ref-type="table" rid="T1">1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Abbreviation of terminologies used in sections Materials and Methods, Results, and Discussion.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Terminology</bold></th>
<th valign="top" align="left"><bold>Abbreviation</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Automated anatomical labeling</td>
<td valign="top" align="left">AAL</td>
</tr>
<tr>
<td valign="top" align="left">Backpropogation</td>
<td valign="top" align="left">BP</td>
</tr>
<tr>
<td valign="top" align="left">Correlation coefficient</td>
<td valign="top" align="left">CC</td>
</tr>
<tr>
<td valign="top" align="left">Cross-validation</td>
<td valign="top" align="left">CV</td>
</tr>
<tr>
<td valign="top" align="left">Deep neural networks</td>
<td valign="top" align="left">DNNs</td>
</tr>
<tr>
<td valign="top" align="left">DNN with the feature selection method</td>
<td valign="top" align="left">DNN-FS</td>
</tr>
<tr>
<td valign="top" align="left">DNN without the feature selection method</td>
<td valign="top" align="left">DNN-woFS</td>
</tr>
<tr>
<td valign="top" align="left">Independent component analysis</td>
<td valign="top" align="left">ICA</td>
</tr>
<tr>
<td valign="top" align="left">Left crus cerebellum</td>
<td valign="top" align="left">LCCB</td>
</tr>
<tr>
<td valign="top" align="left">Left inferior temporal gyrus</td>
<td valign="top" align="left">LITG</td>
</tr>
<tr>
<td valign="top" align="left">Left pars triangularis</td>
<td valign="top" align="left">LPT</td>
</tr>
<tr>
<td valign="top" align="left">Left superior parietal lobule</td>
<td valign="top" align="left">LSPL</td>
</tr>
<tr>
<td valign="top" align="left">Mixed national institute of standards and technology</td>
<td valign="top" align="left">MNIST</td>
</tr>
<tr>
<td valign="top" align="left">National database for autism research</td>
<td valign="top" align="left">NDAR</td>
</tr>
<tr>
<td valign="top" align="left">Neural network</td>
<td valign="top" align="left">NN</td>
</tr>
<tr>
<td valign="top" align="left">Preprocessed connectomes project</td>
<td valign="top" align="left">PCP</td>
</tr>
<tr>
<td valign="top" align="left">Principal component analysis</td>
<td valign="top" align="left">PCA</td>
</tr>
<tr>
<td valign="top" align="left">Regions-of-interest</td>
<td valign="top" align="left">ROI</td>
</tr>
<tr>
<td valign="top" align="left">Right inferior temporal gyrus</td>
<td valign="top" align="left">RITG</td>
</tr>
<tr>
<td valign="top" align="left">Right putamen</td>
<td valign="top" align="left">RP</td>
</tr>
<tr>
<td valign="top" align="left">Right superior frontal gyrus</td>
<td valign="top" align="left">RSFG</td>
</tr>
<tr>
<td valign="top" align="left">Sparse auto-encoders</td>
<td valign="top" align="left">SAE</td>
</tr>
<tr>
<td valign="top" align="left">Time series</td>
<td valign="top" align="left">TS</td>
</tr>
<tr>
<td valign="top" align="left">University of Michigan: Sample 1</td>
<td valign="top" align="left">UM:S1</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec sec-type="materials and methods" id="s2">
<title>Materials and methods</title>
<sec>
<title>Data acquisition and preprocessing</title>
<p>The rs-fMRI dataset used in this paper was obtained from UM:S1 site, ABIDE I (Di Martino et al., <xref ref-type="bibr" rid="B16">2014</xref>). ABIDE I is the first ABIDE initiative, a grassroots consortium aggregating. The UM:S1 dataset contains samples from 110 adolescent subjects among which there are 55 ASD patients and 55 TD controls. Each sample consists of one or more rs-fMRI acquisitions and a volumetric MPRAGE image (Nielsen et al., <xref ref-type="bibr" rid="B53">2013</xref>). The dataset is publicly available at <ext-link ext-link-type="uri" xlink:href="http://fcon_1000.projects.nitrc.org/indi/abide/">http://fcon_1000.projects.nitrc.org/indi/abide/</ext-link>, and all the patient health information associated with the data has been de-identified. The access instructions were detailed in Di Martino et al. (<xref ref-type="bibr" rid="B16">2014</xref>). A diagnosis of ASD was made by applying the Autism Diagnostic Interview Revised (Lord et al., <xref ref-type="bibr" rid="B45">1994</xref>) and the Autism Diagnostic Observation Schedule (Lord et al., <xref ref-type="bibr" rid="B44">2000</xref>), and confirmed by clinical consensus (Lord et al., <xref ref-type="bibr" rid="B43">2006</xref>). Exclusion criteria common to all participants included any history of neurological disorder (including seizures), a history of head trauma, a history of psychosis, a history of bipolar disorder, and if either verbal or non-verbal IQ score was lower than 85. The performance and verbal IQ were obtained using the Peabody Picture Vocabulary Test (Dunn et al., <xref ref-type="bibr" rid="B18">1965</xref>) and the Ravens Progressive Matrices (Lord et al., <xref ref-type="bibr" rid="B45">1994</xref>). Full IQ estimates were based on the average of the performance IQ and verbal IQ scores available and provided in this dataset. More details about this dataset are summarized in Table <xref ref-type="table" rid="T2">2</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Summary of demographics, neuropsychological performance, and clinical characteristics for each of the TD and ASD groups.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th/>
<th valign="top" align="center"><bold>TD controls <italic>N</italic> &#x0003D; 55</bold></th>
<th valign="top" align="center"><bold>ASD patients <italic>N</italic> &#x0003D; 55</bold></th>
<th valign="top" align="center"><bold><italic>t</italic></bold></th>
<th valign="top" align="center"><bold><italic>p</italic>-value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" colspan="5" style="background-color:#bbbdc0"><bold>DEMOGRAPHICS</bold></td>
</tr>
<tr>
<td valign="top" align="left">Age [Mean &#x000B1; <italic>SD</italic> (years)]</td>
<td valign="top" align="center">14.2 &#x000B1; 3.2</td>
<td valign="top" align="center">12.7 &#x000B1; 2.4</td>
<td valign="top" align="center">2.52</td>
<td valign="top" align="center">0.01</td>
</tr>
<tr>
<td valign="top" align="left">Age range [youngest-oldest (years)]</td>
<td valign="top" align="center">8.2&#x02013;19.2</td>
<td valign="top" align="center">8.5&#x02013;18.6</td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">Gender (male/%)</td>
<td valign="top" align="center">38/69.09%</td>
<td valign="top" align="center">46/83.64%</td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">Handness (right/%)<xref ref-type="table-fn" rid="TN1"><sup>&#x0002A;</sup></xref></td>
<td valign="top" align="center">44/80%</td>
<td valign="top" align="center">41/74.55%</td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left" colspan="5" style="background-color:#bbbdc0"><bold>NEUROPSYCHOLOGICAL PERFORMANCE</bold></td>
</tr>
<tr>
<td valign="top" align="left">Verbal IQ (Mean &#x000B1; <italic>SD</italic>)</td>
<td valign="top" align="center">113.49 &#x000B1; 13.72</td>
<td valign="top" align="center">107.04 &#x000B1; 20.34</td>
<td valign="top" align="center">1.93</td>
<td valign="top" align="center">0.06</td>
</tr>
<tr>
<td valign="top" align="left">Performance IQ (Mean &#x000B1; <italic>SD</italic>)<xref ref-type="table-fn" rid="TN1"><sup>&#x0002A;</sup></xref></td>
<td valign="top" align="center">100.96 &#x000B1; 11.58</td>
<td valign="top" align="center">100.06 &#x000B1; 20.08</td>
<td valign="top" align="center">0.28</td>
<td valign="top" align="center">0.78</td>
</tr>
<tr>
<td valign="top" align="left">Full IQ (Mean &#x000B1; <italic>SD</italic>)<xref ref-type="table-fn" rid="TN1"><sup>&#x0002A;</sup></xref></td>
<td valign="top" align="center">106.85 &#x000B1; 9.72</td>
<td valign="top" align="center">103.37 &#x000B1; 17.63</td>
<td valign="top" align="center">1.24</td>
<td valign="top" align="center">0.22</td>
</tr>
<tr>
<td valign="top" align="left" colspan="5" style="background-color:#bbbdc0"><bold>CLINICAL CHARACTERISTICS</bold></td>
</tr>
<tr>
<td valign="top" align="left">ADI-R social (Mean &#x000B1; <italic>SD</italic>)</td>
<td/>
<td valign="top" align="center">19.76 &#x000B1; 4.82</td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">ADI-R verbal (Mean &#x000B1; <italic>SD</italic>)</td>
<td/>
<td valign="top" align="center">15.43 &#x000B1; 3.77</td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left">ADOS social affect (Mean &#x000B1; <italic>SD</italic>)</td>
<td/>
<td valign="top" align="center">8.26 &#x000B1; 3.61</td>
<td/>
<td/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="TN1">
<label>&#x0002A;</label>
<p><italic>Missing values from some subjects were removed in calculating the mean and SD</italic>.</p></fn>
</table-wrap-foot>
</table-wrap>
<p>Imaging was performed on a long bore 3T GE sigma scanner with a 4-channel coil at the University of Michigan&#x00027;s Functional MRI laboratory. For each participant, 300 <inline-formula><mml:math id="M20"><mml:mrow><mml:msubsup><mml:mtext>T</mml:mtext><mml:mn>2</mml:mn><mml:mo>&#x0002A;</mml:mo></mml:msubsup></mml:mrow></mml:math></inline-formula> -weighted BOLD images were collected using a reverse spiral sequence (Glover and Law, <xref ref-type="bibr" rid="B22">2001</xref>). Whole brain coverage was obtained with 40 contiguous 3 mm axial slices (TR &#x0003D; 2,000 ms, TE &#x0003D; 30 ms, flip angle &#x0003D; 90&#x000B0;, FOV &#x0003D; 22 cm, 64 &#x000D7; 64 matrix). Each slice was acquired parallel to the AC-PC line. For the structural images, a high-resolution 3D T1 axial overlay (TR &#x0003D; 8.9, TE &#x0003D; 1.8, flip angle &#x0003D; 15&#x000B0;, FOV &#x0003D; 26 cm, slice thickness &#x0003D; 1.4 mm, 124 slices; matrix &#x0003D; 256 &#x000D7; 160) was acquired for anatomical localization. Additionally, a high-resolution spoiled gradient-recalled acquisition in steady state image acquired sagittally (flip angle &#x0003D; 15&#x000B0;, FOV &#x0003D; 26 cm, slice thickness &#x0003D; 1.4 mm, 110 slices) was used for registration of the functional images (Wiggins et al., <xref ref-type="bibr" rid="B70">2011</xref>).</p>
<p>All preprocessed rs-fMRI data were obtained from the PCP which opens sharing of preprocessed neuroimaging data from ABIDE I. PCP applied four pipelines to preprocess the rs-fMRI data. Processing steps in these pipelines are highly similar. However, the algorithm implementation and parameters used in each step among different pipelines are specific. Although, there is no consensus on the best method for preprocessing rs-fMRI data, it is generally accepted that &#x0201C;scrubbing&#x0201D; the data of motion artifact outliers provides a certain level of protection against this motion-induced bias. The subject&#x00027;s motion produces substantial changes in the timecourses of rs-fMRI data, and can cause systematic but spurious correlation structures throughout the brain. Specifically, many long-distance correlations are decreased by subject motion, whereas many short-distance correlations are increased (Power et al., <xref ref-type="bibr" rid="B56">2012</xref>; Satterthwaite et al., <xref ref-type="bibr" rid="B58">2013</xref>; Yan et al., <xref ref-type="bibr" rid="B72">2013</xref>). These artifacts can distort the strength of FCs, and further affect the diagnosis of all kinds of neurological disorders such as ADHD and ASD (Fair et al., <xref ref-type="bibr" rid="B20">2012</xref>; Alaerts et al., <xref ref-type="bibr" rid="B1">2015</xref>). Among the four pipelines, NIAK includes the &#x0201C;volume scrubbing&#x0201D; step for reducing the head motion effect. In NIAK, the raw rs-fMRI data were preprocessed using the following steps: motion realignment, intensity normalization (non-uniformity correction using median volume), nuisance signal removal, and registration. In the nuisance signal removal step, the scrubbing was employed to clean confounding variation due to physiological processes (heart beat, head motion, and respiration), and head motion. The low frequency scanner drifts from the fMRI signal were cleaned by setting up a discrete cosine basis with a 0.01 Hz high-pass cut-off. The band-pass filtering (0.01&#x02013;0.1 Hz) was applied after nuisance variable regression. In the registration step, the data were spatial normalized to the Montreal Neurological Institute template with 3 mm isotropic voxel size, followed by spatial smoothing using a 6-mm isotropic full-with at half maximum Gaussian kernel. For more data preprocessing details, please refer to Di Martino et al.&#x00027;s paper (Di Martino et al., <xref ref-type="bibr" rid="B16">2014</xref>). The description of pipelines can be found at <ext-link ext-link-type="uri" xlink:href="http://preprocessed-connectomes-project.org/abide/Pipelines.html">http://preprocessed-connectomes-project.org/abide/Pipelines.html</ext-link>.</p>
<p>Neuroscientists at PCP extracted mean time-series for several sets of ROIs atlases, including AAL which can be obtained at <ext-link ext-link-type="uri" xlink:href="http://preprocessed-connectomes-project.org/abide/Pipelines.html">http://preprocessed-connectomes-project.org/abide/Pipelines.html</ext-link>. The mean BOLD TS across voxels in each region of AAL was calculated from each rs-fMRI already registered in standard space. This project used mean BOLD TS from the AAL atlas containing 116 regions. Then Pearson&#x00027;s CCs were calculated using the resulting mean BOLD TS from all 6670 (<inline-formula><mml:math id="M21"><mml:msubsup><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>116</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula>) possible pairs of regions. Figure <xref ref-type="fig" rid="F1">1A</xref> illustrates all steps of getting the whole-brain FCP. However, with the help of PCP, only the final step was needed, i.e., calculating the functional connectivity matrix based on BOLD TS of each voxel in each pair of regions. As with any statistic, Pearson&#x00027;s CCs has a sampling distribution. It will be normally distributed when the absolute value of the correlation in the population is low. However, if high correlation values occur frequently in the population, the distribution has a negative skew effect. To avoid this and force samples to be normally distributed, CCs were Fisher&#x00027;s r-to-z transformed (Rosner, <xref ref-type="bibr" rid="B57">2015</xref>). Then z-scores of each subject were normalized (mean &#x0003D; 0, standard deviation &#x0003D; 1) via pseudo z-scoring. Thus, each subject&#x00027;s measurements were represented as a vector of 6,670 z-scores&#x02014;one for each pair of the 116 brain regions. This vector is designated as the FCP for that subject, and is used as input to the classifiers. Each element of the FCP vector is termed a FC.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The DNN based method for predicting the ASD: <bold>(A)</bold> Functional connectivity (FC) analysis. <bold>(B)</bold> Feature selection based on SAEs. <bold>(C)</bold> Training the DNN. The dashed arrow between <bold>(C)</bold> and <bold>(B)</bold> indicates that the low level SAE in <bold>(C)</bold> comes from <bold>(B)</bold>. Arrows between <bold>(A)</bold> and <bold>(B)</bold>, <bold>(A)</bold> and <bold>(C)</bold> indicate the data flow between modules.</p></caption>
<graphic xlink:href="fnins-11-00460-g0001.tif"/>
</fig>
</sec>
<sec>
<title>Overview of the method</title>
<p>In this paper, a DNN model with a novel feature selection method based on multiple SAEs was proposed for predicting ASD from brain resting-state functional connectivity patterns. The entire model contains three parts: (A) functional connectivity analysis, (B) feature selection based on multiple SAEs, and (C) training the DNN (Figure <xref ref-type="fig" rid="F1">1</xref>). First, each raw rs-fMRI data was preprocessed, and the whole-brain FCP were obtained by calculating the Pearson&#x00027;s CC of TSs from any pair of ROIs. Let <italic>X</italic> &#x02208; <italic>R</italic><sup><italic>N</italic>&#x000D7;<italic>d</italic></sup> denote the training data, in which each row indicates a training sample, and <italic>Y</italic> &#x02208; <italic>R</italic><sup><italic>N</italic></sup> denote their corresponding labels. As discussed above, the dimension of each whole-brain FCP is 6,670 (<italic>d</italic> &#x0003D; 6670). Multiple SAEs were then trained on the training data using gradient descent learning. The hidden neurons of these SAEs after training provided a feature pool. The feature selection algorithm was then used to select features with high discriminating power, which were used to form the first hidden layer of the DNN model (marked by the dashed rectangular box in Figure <xref ref-type="fig" rid="F1">1C</xref>). It is hypothesized that this layer provides low-dimension representations with higher quality compared to representations obtained from a single trained SAE, and such representations would allow the DNN to improve the classification accuracy. Next, several SAEs and an SR model were stacked on top of the constructed feature layer to form a DNN, which was trained (excluding the first hidden layer) by following steps shown in Figure <xref ref-type="fig" rid="F1">1C</xref> using labeled training data.</p>
<p>For training and testing the DNN, and optimizing parameters in the model, the five-fold nested CV framework shown in Figure <xref ref-type="fig" rid="F2">2</xref> was used. After functional connectivity analysis, the obtained whole-brain FCPs were applied as input to the DNN. The DNN classifier was trained and parameters were optimized using training and validation data during the training phase. Then the classification task was performed on each individual in the test data in the test phase to identify the corresponding status (ASD patient or TD control). Finally, the weights of the DNN were analyzed to understand patterns learned by it and to determine ASD-related FCs.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The nested CV framework for the DNN training, testing, and parameters optimization.</p></caption>
<graphic xlink:href="fnins-11-00460-g0002.tif"/>
</fig>
</sec>
<sec>
<title>The novel feature selection method based on multiple sparse auto-encoders</title>
<sec>
<title>Sparse auto-encoders</title>
<p>An AE is a three-layer feed-forward NN as shown in Figure <xref ref-type="fig" rid="F3">3</xref>. It comprises an input layer, a hidden layer, and an output layer. The hidden layer is fully connected to the input layer and the output layer through weighted feed-forward connections. The AE is trained so that its output layer reproduces the stimulus pattern on its input layer. Thus, the number of nodes in the input layer and the output layer is the same, and are both equal to the dimension of the data. Typically, the number of hidden nodes in an AE is much smaller than the number of input and output nodes. To achieve the required reconstruction of its input at the output layer, the AE is forced to infer a maximally information-preserving lower-dimension representation of the input in the hidden layer, which can then be mapped to the output layer. Thus, such an AE performs a dimension reduction function, and each hidden node can be seen as representing a feature of this lower-dimension representation. A convenient way to visualize the feature represented by a hidden node is the pattern of its connection weights from all input nodes (marked in red in Figure <xref ref-type="fig" rid="F3">3</xref>). The AE is called an SAE if a sparsity constraint is imposed on the mean activity of the hidden layer nodes (Larochelle et al., <xref ref-type="bibr" rid="B40">2009</xref>) to reduce overfitting. The number of nodes in the input and output layers is denoted by <italic>d</italic> (each variable of the observation maps to a node) and the number in the hidden layer by <italic>m</italic> (<italic>m</italic> &#x02264; <italic>d</italic>); <italic>W</italic><sup>(1, 0)</sup> &#x02208; <italic>R</italic><sup>(<italic>m</italic>&#x000D7;<italic>d</italic>)</sup> and <italic>W</italic><sup>(2, 1)</sup> &#x02208; <italic>R</italic><sup>(<italic>d</italic>&#x000D7;<italic>m</italic>)</sup>, respectively, denote the encoding weight matrix and the decoding weight matrix; <italic>b</italic><sup>(1, 0)</sup> &#x02208; <italic>R</italic><sup><italic>m</italic></sup> and <italic>b</italic><sup>(2, 1)</sup> &#x02208; <italic>R</italic><sup><italic>d</italic></sup> are, respectively, the bias vectors for the hidden layer and the output layer. The SAE learns a compressed representation of its input by minimizing the reconstruction error between its input pattern and the reconstructed output pattern (Suk et al., <xref ref-type="bibr" rid="B62">2015</xref>).</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The schematic structure of the SAE.</p></caption>
<graphic xlink:href="fnins-11-00460-g0003.tif"/>
</fig>
<p>Let <italic>x</italic> &#x0003D; [<italic>x</italic><sub>1</sub>, <italic>x</italic><sub>2</sub>, &#x02026;<italic>x</italic><sub><italic>w</italic></sub>, &#x02026;<italic>x</italic><sub><italic>d</italic>&#x02212;1</sub>, <italic>x</italic><sub><italic>d</italic></sub>] be an observation from the training data set, <italic>y</italic> &#x0003D; [<italic>y</italic><sub>1</sub>, <italic>y</italic><sub>2</sub>, &#x02026;, <italic>y</italic><sub><italic>h</italic></sub>, &#x02026;<italic>y</italic><sub><italic>n</italic>&#x02212;1</sub>, <italic>y</italic><sub><italic>m</italic></sub>] the compressed hidden layer representation of <italic><bold>x</bold></italic> (<italic>y</italic><sub><italic>h</italic></sub> denotes any hidden layer node between <italic>y</italic><sub>1</sub> and <italic>y</italic><sub><italic>m</italic></sub>), and <italic>z</italic> &#x0003D; [<italic>z</italic><sub>1</sub>, <italic>z</italic><sub>2</sub>, &#x02026;, <italic>z</italic><sub><italic>w</italic></sub>, &#x02026;<italic>z</italic><sub><italic>d</italic>&#x02212;1</sub>, <italic>z</italic><sub><italic>d</italic></sub>] the corresponding output vector. The mapping from <italic><bold>x</bold></italic> to <italic><bold>y</bold></italic> is given by <italic>y</italic> &#x0003D; <italic>f</italic>(<italic>W</italic><sup>(1, 0)</sup><italic>x</italic> &#x0002B; <italic>b</italic><sup>(1, 0)</sup>), where <italic>f</italic> ( ) is the non-linear activation function of the hidden layer nodes. The work reported in this paper uses a logistic sigmoid function (Equation 1) which is widely used in machine learning and pattern recognition tasks (Bengio et al., <xref ref-type="bibr" rid="B4">2007</xref>; Lee et al., <xref ref-type="bibr" rid="B41">2007</xref>; Ngiam et al., <xref ref-type="bibr" rid="B52">2011</xref>; Shin et al., <xref ref-type="bibr" rid="B59">2013</xref>).</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>u</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>/</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x0002B;</mml:mo><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>&#x02212;</mml:mo><mml:mi>u</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>The compressed representation <italic><bold>y</bold></italic> in the hidden layer is mapped to the output <italic><bold>z</bold></italic> by a similar function:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mrow><mml:mi>z</mml:mi><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>W</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mi>y</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mi>b</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>The reconstruction error at the output layer is measured as:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mrow><mml:mi>J</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo>;</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>z</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:msup><mml:mi>z</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mo>&#x0007C;</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:mrow></mml:math></disp-formula>
<p>where <italic>x</italic><sup>(<italic>i</italic>)</sup> is the <italic>i</italic>th input pattern from the training dataset and <italic>z</italic><sup>(<italic>i</italic>)</sup> is the corresponding output of the network for the given input. The network is trained by minimizing a two-term cost function given by:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mrow><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mi>J</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B2;</mml:mi><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>m</mml:mi></mml:munderover><mml:mrow><mml:mi>K</mml:mi><mml:mi>L</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:mo>&#x02032;</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula>
<p>The first term <italic>J</italic>(<italic>W, b</italic>) also has two parts:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mrow><mml:mi>J</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:munderover><mml:mrow><mml:mi>J</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo>;</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mi>&#x003BB;</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>&#x02212;</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>l</mml:mi></mml:msub></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula>
<p>where <italic>M</italic> denotes the total number of observations in the training dataset; <italic>n</italic><sub><italic>l</italic></sub> indicates the number of layers in the network (<italic>n</italic><sub><italic>l</italic></sub> &#x0003D; 3 here); <italic>s</italic><sub><italic>l</italic></sub> denotes the number of nodes in layer <italic>l</italic>; and <inline-formula><mml:math id="M22"><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> is the weight of the connection from node <italic>i</italic> in layer <italic>l</italic>&#x0002B;<italic>1</italic> to unit <italic>j</italic> in layer <italic>l</italic>. The first term represents the reconstruction error averaged over the training set, while the second term&#x02014;called weight decay&#x02014;is a regularization term that tends to decrease the total magnitude of the weights in the SAE and helps prevent overfitting. The parameter &#x003BB; controls the relative importance of the two terms.</p>
<p>The second term in the <italic>Cost</italic> function is also a regularization term called a sparseness constraint. Its purpose is to ensure that only a small number of hidden nodes have significant activity for any given input pattern. Let <italic>y</italic><sub><italic>j</italic></sub>(<italic>x</italic>) denote the activation of the hidden node <italic>j</italic> for input <italic><bold>x</bold></italic>. Then the average activation of the hidden node <italic>h</italic> over the entire training dataset is given by:</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M6"><mml:mrow><mml:mi>&#x003C1;</mml:mi><mml:msub><mml:mo>&#x02032;</mml:mo><mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mo>&#x000A0;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mo>&#x000A0;</mml:mo><mml:mstyle displaystyle='true'><mml:munder><mml:mrow><mml:mover><mml:mo>&#x02211;</mml:mo><mml:mi>M</mml:mi></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:munder></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>A target sparsity parameter &#x003C1; is defined as a small value (e.g., 0.05) such that &#x003C1;&#x02032; <italic>j</italic> is required to be close to &#x003C1; for all hidden units <italic>j</italic>. This means that any individual hidden node must be inactive for most inputs (e.g., active only in about 5% of cases if &#x003C1; &#x0003D; 0.05). This is accomplished by adding the second term to the <italic>Cost</italic> function that penalizes &#x003C1;&#x02032; <italic>j</italic> deviating significantly from &#x003C1;. Since &#x003C1; and &#x003C1;&#x02032; <italic>j</italic> can be seen as the probabilities of Bernoulli random variables to be 1, a Kullback-Leibler (KL) divergence (Shin et al., <xref ref-type="bibr" rid="B59">2013</xref>), denoted by the following formula, is used as the penalty term:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M7"><mml:mrow><mml:mo>&#x000A0;</mml:mo><mml:mi>K</mml:mi><mml:mi>L</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:msub><mml:mo>&#x02032;</mml:mo><mml:mi>j</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:mo>&#x000A0;</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mfrac><mml:mi>&#x003C1;</mml:mi><mml:mrow><mml:mi>&#x003C1;</mml:mi><mml:msub><mml:mo>&#x02032;</mml:mo><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003C1;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003C1;</mml:mi><mml:msub><mml:mo>&#x02032;</mml:mo><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula>
<p>This has the characteristic that <italic>KL</italic>(&#x003C1;||&#x003C1;&#x02032;<italic>j</italic>) &#x0003D; 0 if &#x003C1;&#x02032;<italic>j</italic> &#x0003D; &#x003C1;, and the value increases as &#x003C1;&#x02032; <italic>j</italic> deviates from &#x003C1;. The optimization of the SAE using the <italic>Cost</italic> function in Equation (4) ensures that the average reconstruction error for training data and the deviation of hidden node activity from the target sparsity value are both minimized, with &#x003B2; as the parameter controlling the relative significance of the two objectives. A standard neural network training method called BP (Werbos, <xref ref-type="bibr" rid="B68">1974</xref>) was used in conjunction with an optimization method called limited-memory Broydon-Fletcher-Goldfarb-Shanno optimization (L-BFGS; Liu and Nocedal, <xref ref-type="bibr" rid="B42">1989</xref>) to obtain optimal parameters of the SAE. The parameters of the SAEs, including the target sparsity &#x003C1;, and the objective weighting parameters &#x003BB; and &#x003B2; were set using typical values recommended by other researchers: &#x003C1; &#x0003D; 10<sup>&#x02212;2</sup>, &#x003BB; &#x0003D; 10<sup>&#x02212;4</sup>. Each SAE was trained until the <italic>Cost</italic> function converged.</p>
</sec>
<sec>
<title>The novel feature selection method</title>
<p>As described above, each set of connection weights from the input layer to a specific hidden node define a feature of the trained SAE. The activation of the <italic>ith</italic> (0 &#x0003C; <italic>i</italic> &#x02264; <italic>m</italic>) hidden layer node is given by Equation (8), where <inline-formula><mml:math id="M23"><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> denotes the connection weight from node <italic>j</italic> in the input layer to node <italic>i</italic> in the hidden layer, <italic>p</italic><sub><italic>j</italic></sub> denotes the <italic>jth</italic> component in the whole-brain FCP, <inline-formula><mml:math id="M24"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula> denotes the bias for hidden node <italic>i</italic>, and <italic>f</italic> is the logistic sigmoid function:</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M8"><mml:mrow><mml:msub><mml:mi>a</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:msub><mml:mi>p</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:mstyle><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:math></disp-formula>
<p>This can be seen as a filter tuned to a particular input pattern <italic><bold>x</bold></italic> which maximizes <italic>a</italic><sub><italic>i</italic></sub>. This pattern can be specified uniquely by constraining the norm of the input whole-brain FCPs as:</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M10"><mml:mrow><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:mi>x</mml:mi><mml:mo>&#x0007C;</mml:mo><mml:msup><mml:mo>&#x0007C;</mml:mo><mml:mn>2</mml:mn></mml:msup><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mi>p</mml:mi><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>&#x02264;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula>
<p>Then the input pattern that maximally activates hidden node <italic>i</italic> is given by setting the FC <italic>p</italic><sub><italic>j</italic></sub> (for all <italic>n</italic> FCs, <italic>j</italic> &#x0003D; 1,&#x02026;,<italic>n</italic>) to:</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M9"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>n</mml:mi></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula>
<p>which defines the pattern of whole-brain FCP which evokes a maximal response from hidden node <italic>a</italic><sub><italic>i</italic></sub> and is, therefore, the feature defined by that node.</p>
<p>Many features of the whole-brain FCP can be detected by a single trained SAE, e.g., an SAE with 200 hidden layer nodes can learn 200 features of whole-brain FCP, each of which fires the corresponding node maximally. However, redundant features often emerge in the feature layer, so that many hidden layer nodes have very similar activation values, reducing the representational range of the hidden layer. Typically, even with regularization, the number of diverse features in a single trained SAE after excluding redundant features is not enough for a sufficiently informative compressed representation. The proposed novel feature selection method based on multiple SAEs can address this problem by selecting a number of diverse features with high discriminating power (ASD vs. TD) from a large diverse but redundant feature pool. The accuracy of the classification model trained by such representations from the new feature layer should improve as a result. Figure <xref ref-type="fig" rid="F4">4</xref> illustrates the steps of the proposed feature selection method and the associated algorithm in detail.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The novel feature selection method: <bold>(A)</bold> Feature selection steps. <bold>(B)</bold> The feature selection algorithm.</p></caption>
<graphic xlink:href="fnins-11-00460-g0004.tif"/>
</fig>
<p>As Figure <xref ref-type="fig" rid="F4">4A</xref> illustrates, first of all, <italic>L</italic> (<italic>L</italic> &#x0003D; 15 in this study) SAEs, each with <italic>m</italic> hidden layer nodes, are trained on the three-fold training dataset as shown in Figure <xref ref-type="fig" rid="F2">2</xref>. <italic>L</italic>&#x02014;the number of SAEs used to generate the feature pool&#x02014;is an important parameter. Using a larger value of <italic>L</italic> results in a larger, but possibly more redundant, feature pool. Using a smaller L produces a smaller feature pool with less feature diversity, which would cause degraded performance. In our study, we actually tried different values of <italic>L</italic> (5, 10, 15, 20) to decide the appropriate number, and it turned out that the deep learning framework associated with the proposed feature selection method obtained the best result in the ASD vs. TD classification task when <italic>L</italic> is equal to 15. The <italic>L</italic>&#x000D7;<italic>m</italic> features collected from these form a diverse, redundant feature pool. The next step of the feature selection algorithm is to determine the discriminating power of each feature in the pool by calculating its Fisher&#x00027;s score (Weston et al., <xref ref-type="bibr" rid="B69">2000</xref>), which can be defined as Equation (11):</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M11"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>E</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02212;</mml:mo><mml:msubsup><mml:mi>E</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>&#x003C3;</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>&#x003C3;</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Here, <inline-formula><mml:math id="M25"><mml:msubsup><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is the mean activation value of the <italic>i</italic>th hidden layer node for all inputs in the TD group, i.e.,</p>
<disp-formula id="E12"><label>(12)</label><mml:math id="M12"><mml:mrow><mml:msubsup><mml:mi>E</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munderover><mml:mrow><mml:msubsup><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:msubsup></mml:mrow></mml:mstyle><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>d</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mi>s</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mi>j</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula>
<p>where <italic>sgm</italic>() is the sigmoid function; <inline-formula><mml:math id="M26"><mml:mrow><mml:msubsup><mml:mi>&#x003C3;</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mi>N</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>a</mml:mi><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:msubsup><mml:mo>&#x02212;</mml:mo><mml:msubsup><mml:mi>E</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>T</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:math></inline-formula> is the variance of the hidden node input for the TD group; <italic>N</italic><sup><italic>TD</italic></sup> and <italic>N</italic><sup><italic>ASD</italic></sup> are the number of subjects in the TD control and ASD groups, respectively; and <inline-formula><mml:math id="M27"><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> is the connection weight to the <italic>i</italic>th node in the hidden layer from the <italic>j</italic>th node in the input layer. <inline-formula><mml:math id="M28"><mml:msubsup><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> and <inline-formula><mml:math id="M29"><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:mi>S</mml:mi><mml:mi>D</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> are defined similarly. The feature is considered to have high discriminating power (ASD vs. TD) when its Fisher&#x00027;s score is high. Features with high discriminating power are of interest during the feature selection procedure due to the fact that they can generate more informative representations for the classification model. For selection, features in the pool are sorted by their Fisher&#x00027;s scores in descending order. Then a number of dissimilar features from the pool with high discriminating power are selected. In order to choose diverse features, a Euclidean distance similarity metric is applied to measure the similarity of a pair of features. Figure <xref ref-type="fig" rid="F4">4B</xref> illustrates the procedure for feature selection in detail. <italic>feature</italic>_<italic>number</italic> indicates the number of features to be selected from the pool, and <italic>similarity</italic>_<italic>threshold</italic> indicates a threshold for the feature pair similarity. The algorithm considers features in descending order of discriminating power. The criterion for selection is that a feature should be selected only if its similarity to any other features in the set <bold>F</bold> of already selected features is below the <italic>similarity_threshold</italic> value. Otherwise, the next feature is considered. This procedure is not terminated until the number of selected features is equal to <italic>feature</italic>_<italic>number</italic>, or all features in the pool have been examined. Finally, a diverse and highly discriminating feature layer is re-constructed using features collected in <bold>F</bold>.</p>
</sec>
<sec>
<title>Training the DNN model</title>
<p>The entire DNN training process comprises three steps as illustrated in Figure <xref ref-type="fig" rid="F1">1C</xref>, including unsupervised training of the stacked SAEs, supervised training of the SR model, and fine tuning of the entire DNN. The first two steps are considered to be pre-training of the DNN model. The DNN is a multilayer NN constructed by stacking multiple SAEs and the SR model together. The stacked SAEs are used to do successive data dimension reductions, and the final reduced-dimension representation is provided as input to the SR model. In the DNN, weights in lower layers are difficult to update by the standard back-propagation algorithm. This is because the algorithm involves the backward propagation of objective function gradients from upper layers to lower layers, and these vanish quickly as the depth of the network increases (Hinton, <xref ref-type="bibr" rid="B27">2007</xref>). As a result, weights in lower layers change slowly, causing the lower layers to not learn much. A layer-wise pre-training method (Hinton et al., <xref ref-type="bibr" rid="B28">2006</xref>) can greatly improve the training process of a multi-layer NN, and is widely used in various DNN-based applications (Suk et al., <xref ref-type="bibr" rid="B61">2014</xref>, <xref ref-type="bibr" rid="B62">2015</xref>; Kim et al., <xref ref-type="bibr" rid="B38">2016</xref>). This is the approach used to train the current model. Figure <xref ref-type="fig" rid="F5">5</xref> shows the three SAEs used to form the stacked SAEs and illustrates the layer-wise training process. The lowest SAE was first trained with the training dataset (Figure <xref ref-type="fig" rid="F5">5A</xref>). Then each image <italic>x</italic> was used as input to the trained SAE to get the corresponding hidden layer activation vectors <italic>h</italic><sup>(1)</sup> that are considered to represent latent features of the data. After collecting this representation for each image, the middle SAE was trained using these collected representations as input as desired output, as Figure <xref ref-type="fig" rid="F5">5B</xref> illustrates. Then the hidden layer activation vectors <italic>h</italic><sup>(2)</sup> for each data point were collected from this SAE and used as inputs to train the next SAE. This process is shown in Figure <xref ref-type="fig" rid="F5">5C</xref>. Finally, the outputs of the third SAE&#x00027;s hidden layer, <italic>h</italic><sup>(3)</sup>, that are considered as representing the most complex non-linear features latent in the raw data, were generated in the same way as <italic>h</italic><sup>(1)</sup> and <italic>h</italic><sup>(2)</sup>. The output of <italic>h</italic><sup>(3)</sup> was then used as input to train the SR model. The data label information was not involved in the entire layer-wise training process of the stacked SAEs, so it was considered to be unsupervised training. The one exception is the training of the lowest level SAE, which uses the feature selection algorithm and uses labeled data to calculate the discrimination power of features. The SR model is trained by both training data representations and the corresponding labels so the training process is supervised. The pre-training is able to provide a reasonable initialization of parameters for the fine-tuning step (Erhan et al., <xref ref-type="bibr" rid="B19">2010</xref>) so that parameters can be adjusted quickly according to training data labels in a few training iterations.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>The process of layer-wise training the stacked SAEs. <bold>(A)</bold> The first SAE. <bold>(B)</bold> The second SAE. <bold>(C)</bold> The Third SAE.</p></caption>
<graphic xlink:href="fnins-11-00460-g0005.tif"/>
</fig>
<p>The ultimate goal of training the DNN is to construct a diagnosis model that distinguishes ASD patients from TD controls. In the fine-tuning step, all training data associated with labels were used to train the DNN. The BP algorithm in conjunction with L-BFGS was applied to optimize all the weights. The cost function of the DNN is defined by Equation (13) and Equation (14).</p>
<disp-formula id="E13"><label>(13)</label><mml:math id="M13"><mml:mrow><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:munderover><mml:mrow><mml:mi>J</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo>;</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mstyle><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mfrac><mml:mi>&#x003BB;</mml:mi><mml:mn>2</mml:mn></mml:mfrac><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mo>&#x02212;</mml:mo><mml:mn>2</mml:mn></mml:mrow></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo>=</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>l</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>s</mml:mi><mml:mi>l</mml:mi></mml:msub></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula>
<disp-formula id="E14"><label>(14)</label><mml:math id="M14"><mml:mrow><mml:mi>J</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo>;</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mo>&#x0007C;</mml:mo><mml:mo>&#x0007C;</mml:mo><mml:msub><mml:mi>h</mml:mi><mml:mrow><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi></mml:mrow></mml:msub><mml:mo stretchy='false'>(</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>&#x0007C;</mml:mo><mml:msubsup><mml:mo>&#x0007C;</mml:mo><mml:mn>2</mml:mn><mml:mn>2</mml:mn></mml:msubsup></mml:mrow></mml:math></disp-formula>
<p>In Equation (13), <italic>M</italic> denotes the total number of subjects in the training dataset, <italic>W</italic> denotes the weights parameters in the DNNs, <italic>b</italic> denotes the bias vector, <italic>x</italic><sup>(<italic>k</italic>)</sup> denotes the input for <italic>k</italic>th subject in the training dataset, <italic>y</italic><sup>(<italic>k</italic>)</sup> is its corresponding label which can indicate the condition status (ASD 1, TD control 0), <italic>n</italic><sub><italic>l</italic></sub> indicates the total number of layers of the DNNs, <italic>s</italic><sub><italic>l</italic></sub> indicates the total number of nodes in the layer <italic>l</italic>, <inline-formula><mml:math id="M30"><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> presents the weight between <italic>i</italic>th node in layer <italic>l</italic>&#x0002B;1 and the <italic>j</italic>th node in the layer <italic>l</italic>. From Equation (4), it can be seen that <italic>J</italic>(<italic>W, b</italic>; <italic>x</italic><sup>(<italic>i</italic>)</sup>, <italic>z</italic><sup>(<italic>i</italic>)</sup>) measures the error between the predicted label of a certain subject <inline-formula><mml:math id="M31"><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>W</mml:mi><mml:mo>,</mml:mo><mml:mi>b</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> and the corresponding real label <italic>y</italic><sup>(<italic>k</italic>)</sup>. Based on the cost function, the weights optimization process can be described as following::</p>
<disp-formula id="E15"><label>(15)</label><mml:math id="M15"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>[</mml:mo><mml:mo stretchy='false'>(</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mi>&#x003B4;</mml:mi><mml:msup><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>&#x003BB;</mml:mi><mml:msup><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>]</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E16"><label>(16)</label><mml:math id="M16"><mml:msup><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>&#x02212;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>[</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>M</mml:mi></mml:mfrac><mml:mi>&#x003B4;</mml:mi><mml:msup><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo stretchy='false'>]</mml:mo></mml:math></disp-formula>
<disp-formula id="E17"><label>(17)</label><mml:math id="M17"><mml:mi>&#x003B4;</mml:mi><mml:msup><mml:mrow><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:munderover><mml:mrow><mml:mfrac><mml:mo>&#x02202;</mml:mo><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle><mml:mi>J</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>W</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>;</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:math></disp-formula>
<disp-formula id="E18"><label>(18)</label><mml:math id="M18"><mml:mrow><mml:mi>&#x003B4;</mml:mi><mml:msup><mml:mrow><mml:msub><mml:mi>b</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>M</mml:mi></mml:munderover><mml:mrow><mml:mfrac><mml:mo>&#x02202;</mml:mo><mml:mrow><mml:mo>&#x02202;</mml:mo><mml:msubsup><mml:mi>b</mml:mi><mml:mi>j</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle><mml:mi>J</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>W</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>,</mml:mo><mml:mi>b</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>;</mml:mo><mml:msup><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>k</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:math></disp-formula>
<p>where &#x003B4;<italic>W</italic><sup>(<italic>l</italic>&#x0002B;1, <italic>l</italic>)</sup>(<italic>t</italic>) is the sum of partial derivatives of the cost function with respect to <italic>W</italic><sup>(<italic>l</italic>&#x0002B;1, <italic>l</italic>)</sup>(<italic>t</italic>); &#x003B4;<italic>b</italic><sup>(<italic>l</italic>&#x0002B;1, <italic>l</italic>)</sup>(<italic>t</italic>) is the sum of partial derivatives of the cost function with respect to <inline-formula><mml:math id="M32"><mml:msubsup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>, and <italic>t</italic> is the current iteration number. In this work, all of the training data was used to update parameters during each epoch until the cost function converged. The weight decay parameter &#x003BB; was fixed to 10<sup>&#x02212;5</sup> to prevent overfitting of weights (Moody et al., <xref ref-type="bibr" rid="B51">1995</xref>), the learning rate was initially set to 0.0015 and then gradually reduced after 200 epochs (Darken and Moody, <xref ref-type="bibr" rid="B13">1990</xref>). The number of training epochs was 400.</p>
</sec>
<sec>
<title>The learnt DNN features</title>
<p>FCPs from all 110 subjects were used to train the DNNs once rather than using the nested CV scheme illustrated in Figure <xref ref-type="fig" rid="F2">2</xref>. This training scheme can save the complication of merging DNN weights obtained from different training sets although the classification accuracy is not available simultaneously. The sparsity parameter was set to the value selected most frequently for the final test during the nested CV scheme.</p>
<p>Analyzing the weights from the input layer to hidden nodes and between successive layers of hidden nodes can help in explaining the basis of classification learned by DNNs. Features in DNNs trained on the MNIST dataset (Haykin and Kosko, <xref ref-type="bibr" rid="B25">2001</xref>) have been shown to correspond to simple image components such as edges by analyzing the weights between the input and the first hidden layers, and the linear combination of these features to the second layer can be seen as detecting more complex elements such as corners. An approach similar to these and other studies (Denil et al., <xref ref-type="bibr" rid="B14">2013</xref>; Suk et al., <xref ref-type="bibr" rid="B61">2014</xref>; Kim et al., <xref ref-type="bibr" rid="B38">2016</xref>) was used to analyze the feature for each hidden layer node in each hidden layer in the DNNs. The feature vector of <italic>i</italic>th node in the (<italic>l</italic> &#x0002B; 1)th layer is specified as:</p>
<disp-formula id="E19"><label>(19)</label><mml:math id="M19"><mml:mrow><mml:msubsup><mml:mi>F</mml:mi><mml:mi>i</mml:mi><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>Z</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup><mml:msubsup><mml:mi>F</mml:mi><mml:mi>j</mml:mi><mml:mi>l</mml:mi></mml:msubsup></mml:mrow></mml:mstyle><mml:mtext>&#x000A0;and&#x000A0;</mml:mtext><mml:msubsup><mml:mi>F</mml:mi><mml:mi>i</mml:mi><mml:mn>1</mml:mn></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mi>W</mml:mi><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mo>:</mml:mo><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy='false'>)</mml:mo></mml:mrow></mml:msubsup></mml:mrow></mml:math></disp-formula>
<p>Here, <inline-formula><mml:math id="M33"><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>l</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> is the weight between the <italic>i</italic>th node in the (<italic>l</italic> &#x0002B; 1)th layer and the <italic>j</italic>th node in the layer <italic>l</italic>. To define the feature for each hidden node, the <italic>Z</italic> connections with the highest weight magnitudes between (<italic>l</italic> &#x0002B; 1)th and <italic>l</italic>th layer were chosen, and used in the analysis of features. For each node in the first hidden layer, the weights between itself and all input nodes are considered as the learnt feature. So <inline-formula><mml:math id="M34"><mml:msubsup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mo>:</mml:mo></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msubsup></mml:math></inline-formula> which is actually <inline-formula><mml:math id="M35"><mml:msubsup><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula>, indicates all weights of connections connecting <italic>i</italic>th node in the first hidden layer to all nodes from the input layer. Figure <xref ref-type="fig" rid="F6">6</xref> illustrates the feature construction procedure for a hidden layer node in the third layer. It can be seen that a feature existing in the higher hidden layer of the DNN is actually the linear combination of features from lower hidden layers. Because the DNN features have the hierarchical property, a top-down ASD related biomarker identification method can be developed. The detailed procedure and the result will be detailed in Section Visualization of Significant FCs Identified by Learning.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>An example of the representation of learnt features from DNN (Connections used to construct features are presented by solid arrows).</p></caption>
<graphic xlink:href="fnins-11-00460-g0006.tif"/>
</fig>
</sec>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec>
<title>Comparing the performance of the DNN with and without the feature selection method</title>
<p>To assess the benefit of the proposed feature selection method, the performance of the DNN-FS illustrated in Figure <xref ref-type="fig" rid="F1">1</xref> was compared with that of the DNN-woFS. In DNN-FS, the weights between the first hidden layer and the input layer of the stacked SAEs were selected by the proposed feature selection method, and were trained further during the 3-step DNN training. All other weights were initialized randomly. The training proceeded from the second hidden layer onward, i.e., weights to the second hidden layer from the first were trained first, then those from the second to the third, and so on). The reduced-dimension representations of the whole-brain FCPs from the first hidden layer were considered as the input to the network from the second hidden layer on. In DNN-woFS, all weights were randomly initialized and training proceeded from the first hidden layer with the whole-brain FCPs were the input. The hypothesis was that the stacked SAEs in DNN-FS would generate reduced-dimension, highly discriminative representations with higher quality than those generated by DNN-woFS, and that, after training, this would be visible in the classification accuracy of the SR classifiers using each system. Several comparison scenarios were designed to test the hypothesis. In each scenario, DNN-FS and DNN-woFS had the same configuration in the number of hidden layers and the number of hidden layer nodes. This configuration varied among different scenarios so that the performance of the systems could be evaluated with different architectures. The evaluation scheme for both models in each scenario was the five-fold CV, and both used the parameter settings discussed in Section Sparse Auto-Encoders and The Novel Feature Selection Method.</p>
<p>Tables <xref ref-type="table" rid="T3">3</xref>, <xref ref-type="table" rid="T4">4</xref> present the classification accuracy from DNN-FS and DNN-woFS, respectively in different scenarios. From Table <xref ref-type="table" rid="T3">3</xref> it is clear that, for the DNN-woFS, the best performance was 81.82% when it had three hidden layers, each of which had 100 nodes. The best performance in Table <xref ref-type="table" rid="T4">4</xref> was 86.36% coming from the DNN-FS with three layers, each of which had 150 hidden nodes. From both tables, it can also be observed that simply increasing the number of hidden layer nodes does not always help improve accuracy when the number of hidden layers stays the same. Similarly, simply increasing the number of layers does not always help the stacked SAEs generate more robust representations when the number of nodes is fixed.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Results from the DNN-woFS in different scenarios.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>No. of nodes</bold></th>
<th valign="top" align="center" colspan="5" style="border-bottom: thin solid #000000;"><bold>No. of Layers</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>1 (%)</bold></th>
<th valign="top" align="center"><bold>2 (%)</bold></th>
<th valign="top" align="center"><bold>3 (%)</bold></th>
<th valign="top" align="center"><bold>4 (%)</bold></th>
<th valign="top" align="center"><bold>5 (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">50</td>
<td valign="top" align="center">56.36</td>
<td valign="top" align="center">61.82</td>
<td valign="top" align="center">70.91</td>
<td valign="top" align="center">66.36</td>
<td valign="top" align="center">67.27</td>
</tr>
<tr>
<td valign="top" align="left">100</td>
<td valign="top" align="center">63.64</td>
<td valign="top" align="center">71.82</td>
<td valign="top" align="center"><bold>81.82</bold></td>
<td valign="top" align="center">77.27</td>
<td valign="top" align="center">74.55</td>
</tr>
<tr>
<td valign="top" align="left">150</td>
<td valign="top" align="center">65.45</td>
<td valign="top" align="center">70.00</td>
<td valign="top" align="center">77.27</td>
<td valign="top" align="center">73.63</td>
<td valign="top" align="center">70.91</td>
</tr>
<tr>
<td valign="top" align="left">200</td>
<td valign="top" align="center">68.18</td>
<td valign="top" align="center">70.91</td>
<td valign="top" align="center">76.36</td>
<td valign="top" align="center">78.18</td>
<td valign="top" align="center">76.36</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>The best performance is marked with bold</italic>.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Results from the DNN-FS in different scenarios.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>No. of nodes</bold></th>
<th valign="top" align="center" colspan="5" style="border-bottom: thin solid #000000;"><bold>No. of Layers</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>1 (%)</bold></th>
<th valign="top" align="center"><bold>2 (%)</bold></th>
<th valign="top" align="center"><bold>3 (%)</bold></th>
<th valign="top" align="center"><bold>4 (%)</bold></th>
<th valign="top" align="center"><bold>5 (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">50</td>
<td valign="top" align="center">59.09</td>
<td valign="top" align="center">64.55</td>
<td valign="top" align="center">72.73</td>
<td valign="top" align="center">67.27</td>
<td valign="top" align="center">69.09</td>
</tr>
<tr>
<td valign="top" align="left">100</td>
<td valign="top" align="center">67.27</td>
<td valign="top" align="center">75.45</td>
<td valign="top" align="center">85.45</td>
<td valign="top" align="center">81.82</td>
<td valign="top" align="center">77.27</td>
</tr>
<tr>
<td valign="top" align="left">150</td>
<td valign="top" align="center">68.18</td>
<td valign="top" align="center">78.18</td>
<td valign="top" align="center"><bold>86.36</bold></td>
<td valign="top" align="center">75.45</td>
<td valign="top" align="center">72.73</td>
</tr>
<tr>
<td valign="top" align="left">200</td>
<td valign="top" align="center">69.09</td>
<td valign="top" align="center">72.73</td>
<td valign="top" align="center">77.27</td>
<td valign="top" align="center">79.09</td>
<td valign="top" align="center">78.18</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>The best performance is marked with bold</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>The results in Tables <xref ref-type="table" rid="T3">3</xref>, <xref ref-type="table" rid="T4">4</xref> provide strong support for the hypothesis that the feature selection method used in DNN-FS networks helps the stacked SAEs generate more competitive representations for the ASD diagnosis task. Compared with the results from DNN-woFS, DNN-FS classification accuracy was always higher in every comparison scenario (Table <xref ref-type="table" rid="T5">5</xref>). The color scale green-yellow-red corresponds to low-medium-high improvement, while the numerical values indicate the exact percentage improvement. The most significant improvement (9.09%) occurs in the DNN-FS with three hidden layers, each of which has 150 nodes. However, the improvement of the DNN-FS is low when each hidden layer contains 200 nodes, even when the number of hidden layers is varied.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>The improvement of the DNN-FS in different scenarios compared with the DNN-woFS.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>No. of nodes</bold></th>
<th valign="top" align="center" colspan="5" style="border-bottom: thin solid #000000;"><bold>No. of layers</bold></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>1 (%)</bold></th>
<th valign="top" align="center"><bold>2 (%)</bold></th>
<th valign="top" align="center"><bold>3 (%)</bold></th>
<th valign="top" align="center"><bold>4 (%)</bold></th>
<th valign="top" align="center"><bold>5 (%)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">50</td>
<td valign="top" align="center" style="background-color:#fce485">2.73</td>
<td valign="top" align="center" style="background-color:#fce485">2.73</td>
<td valign="top" align="center" style="background-color:#cbdb81">1.82</td>
<td valign="top" align="center" style="background-color:#63bd7b">0.91</td>
<td valign="top" align="center" style="background-color:#cbdb81">1.82</td>
</tr>
<tr>
<td valign="top" align="left">100</td>
<td valign="top" align="center" style="background-color:#ffd27f">3.63</td>
<td valign="top" align="center" style="background-color:#ffd27f">3.63</td>
<td valign="top" align="center" style="background-color:#ffd27f">3.63</td>
<td valign="top" align="center" style="background-color:#fabf7c">4.55</td>
<td valign="top" align="center" style="background-color:#fce485">2.72</td>
</tr>
<tr>
<td valign="top" align="left">150</td>
<td valign="top" align="center" style="background-color:#fce485">2.73</td>
<td valign="top" align="center" style="background-color:#f1796e">8.18</td>
<td valign="top" align="center" style="background-color:#f1796e">9.09</td>
<td valign="top" align="center" style="background-color:#cbdb81">1.82</td>
<td valign="top" align="center" style="background-color:#cbdb81">1.82</td>
</tr>
<tr>
<td valign="top" align="left">200</td>
<td valign="top" align="center" style="background-color:#63bd7b">0.91</td>
<td valign="top" align="center" style="background-color:#cbdb81">1.82</td>
<td valign="top" align="center" style="background-color:#63bd7b">0.91</td>
<td valign="top" align="center" style="background-color:#63bd7b">0.91</td>
<td valign="top" align="center" style="background-color:#cbdb81">1.82</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>Analysis of layer-wise discriminative power</title>
<p>To quantitatively determine the discriminative power of each hidden layer in the DNN-FS and the DNN-woFS, DNN-FS and DNN-woFS networks with the same architecture were trained on FCPs from all subjects, and the discriminative powers of the corresponding layers were determined by calculating the mean Fisher&#x00027;s score and its standard deviation for each layer. Since the DNN-FS with three hidden layers with 150 nodes each generated the best performance among all scenarios, the analysis was conducted on networks with this architecture.</p>
<p>The results showed that the higher layer always have higher mean Fisher&#x00027;s score in both DNN-FS and DNN-woFS (Figure <xref ref-type="fig" rid="F7">7</xref>). This demonstrates that the hierarchical property of the DNN can help it improve its discriminative power for classifying the two groups (ASD patients/TD controls) in a systematic manner, and multiple hidden layers are necessary. It was also observed that the DNN-FS has stronger discriminative power in each layer compared with DNN-woFS. Clearly, the proposed feature selection method improves the DNN classification accuracy by generating highly discriminative representations.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>The mean Fisher&#x00027;s score and its <italic>SD</italic> (mean/<italic>SD</italic>) for each hidden layer from the DNN-FS and the DNN-woFS. <bold>(A)</bold> The first hidden layer. <bold>(B)</bold> The second hidden layer. <bold>(C)</bold> The third hidden layer.</p></caption>
<graphic xlink:href="fnins-11-00460-g0007.tif"/>
</fig>
</sec>
<sec>
<title>Comparing the performance of the stacked SAEs in DNN-FS with other feature selection methods</title>
<p>The work reported in this paper used stacked SAEs for dimension reduction in the DNN-FS framework. The proposed feature selection method helps the stacked SAEs generate reduced-dimension representations with high quality. Training with such representations, the classification accuracy of the SR model is improved significantly. To compare the data dimension reduction performance of the stacked SAEs in the DNN-FS with the performance of other methods, two other feature selection methods&#x02014;two sample <italic>t</italic>-test and elastic net (Zou and Hastie, <xref ref-type="bibr" rid="B75">2005</xref>)&#x02014;were also implemented and tested, using individual FCs as the pool of features. The low-dimension representations generated by each method were used to train SR classifiers, and the performance of these classifiers as well as the DNN-FS-based SR classifier was evaluated on the test dataset. The training scheme for all three SR models was five-fold CV, and the training data was the same for them in each fold. As discussed above, the proposed feature selection method allowed a DNN with three layers and 150 nodes per hidden layer to generate the best performance, so this was the DNN architecture used here. In the two sample <italic>t</italic>-test, the distributions of each FC for the ASD and TD sets were compared, and the 150 FCs with the lowest <italic>p</italic>-values were selected out of 6,670 connections in each fold. The mean FCPs for the ASD and TD groups, and the 150 features with the lowest p-values are shown in Figure <xref ref-type="fig" rid="F8">8</xref>. The elastic net was also applied to select 150 FCs from the 6,670 FCs. To obtain reliable results, each method was run 10 times. In each trial, subjects were randomly divided into five-folds. The average accuracy as well as the standard deviation from each method is shown in Figure <xref ref-type="fig" rid="F9">9</xref>. As the figure illustrates, the DNN-FS method outperformed the other two methods, producing more robust, better quality low-dimensional representations for the SR model to classify TD controls.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>The visualization of group FCs. <bold>(A)</bold> Mean FCs of TD group (&#x02212;0.2 &#x02264; <italic>r</italic> &#x02264; 1). <bold>(B)</bold> Mean FCs of ASD group (&#x02212;0.2 &#x02264; <italic>r</italic> &#x02264; 1). <bold>(C)</bold> Group differences in the 150 FC patterns (evaluated by a two sample <italic>t</italic>-test with a threshold of <italic>p</italic> &#x0003C; 0.0014). Both x and y axes of each subfigure indicate areas in AAL atlas (SC, subcortical area, CB, cerebellum).</p></caption>
<graphic xlink:href="fnins-11-00460-g0008.tif"/>
</fig>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>The comparison of three feature selection methods.</p></caption>
<graphic xlink:href="fnins-11-00460-g0009.tif"/>
</fig>
</sec>
<sec>
<title>Visualization of significant FCs identified by learning</title>
<p>As discussed earlier, the DNNs learn patterns from FCPs as features in a hierarchical manner: a feature in a higher layer is the linear combination of features from the layer below it. Thus, it is possible to look at what features each layer has learned, and use them to infer what elements of the FCPs were most important for classification and, therefore, for the diagnosis of ASD. Applying this approach, important FCs with significant discriminative power were identified, and visualized in the circular graph. More specifically, after conducting the experiment described in Section Analysis of Layer-Wise Discriminative Power, the four nodes with the largest Fisher&#x00027;s scores were located in the third hidden layer of the DNN-FS-based network with the best performance. For each of these, the two nodes with the largest connection weights from the second hidden layer were selected, giving up to eight second hidden layer nodes that contribute most to important features in the third hidden layer. The same process was iteratively applied on the first hidden layer and the input layer, resulting in 32 FCs which made high contribution to discriminate ASD patients from TD controls. These 32 FC elements are visualized in Figure <xref ref-type="fig" rid="F10">10</xref>. Furthermore, for purposes of network analysis, they are summarized in Table <xref ref-type="table" rid="T6">6</xref>, with both terminal brain areas of each identified FC assigned to a brain functional network according to the literature. A few terminal areas that are not in any network defined in the extant research, i.e., the DM network, the FP network, the CO network and the CB network, are left blank. For each included FC, the mean <italic>r</italic> value for each group as well as the group difference was calculated, and the <italic>p</italic>-value obtained by the two sample <italic>t</italic>-test used to evaluate the significance. The significance threshold for discriminative power is <italic>p</italic> &#x0003C; 0.05. The FC was marked in green when its terminal brain regions were significantly positively correlated, and in red when they were significantly negatively correlated. All brain regions in the table were named by AAL labels (Tzourio-Mazoyer et al., <xref ref-type="bibr" rid="B65">2002</xref>).</p>
<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>The visualization of 32 identified FC elements. <bold>(A)</bold> The circular visualization. Thicker connections indicate regions are strongly correlated, and vice versa. &#x0201C;circularGraph&#x0201D; toolbox (<ext-link ext-link-type="uri" xlink:href="http://www.mathworks.com/matlabcentral">www.mathworks.com/matlabcentral</ext-link>) was applied to draw the figure. <bold>(B)</bold> The axial visualization in AAL atlas. Labels information was from AAL atlas. The thicker connection indicates two regions are strongly correlated, and vice versa. The BrainNet Viewer software (<ext-link ext-link-type="uri" xlink:href="http://www.nitrc.org/projects/bnv">www.nitrc.org/projects/bnv</ext-link>) was applied to draw the figure.</p></caption>
<graphic xlink:href="fnins-11-00460-g0010.tif"/>
</fig>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>The network analysis of 32 most significant FC elements.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Connection ID</bold></th>
<th valign="top" align="left"><bold>Regions</bold></th>
<th valign="top" align="center"><bold>Network</bold></th>
<th valign="top" align="center"><bold>Mean CC in ASD group</bold></th>
<th valign="top" align="center"><bold>Mean CC in TDC group</bold></th>
<th valign="top" align="center"><bold>Mean difference</bold></th>
<th valign="top" align="center"><bold><italic>P</italic>-value</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left" style="background-color:#28af57">1</td>
<td valign="top" align="left" style="background-color:#28af57">(4) Frontal_Sup_R</td>
<td valign="top" align="center" style="background-color:#28af57">CO</td>
<td valign="top" align="center" style="background-color:#28af57">0.86</td>
<td valign="top" align="center" style="background-color:#28af57">0.59</td>
<td valign="top" align="center" style="background-color:#28af57">0.27</td>
<td valign="top" align="center" style="background-color:#28af57">0.007</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(30) Insula_R</td>
<td valign="top" align="center" style="background-color:#28af57">CO</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">2</td>
<td valign="top" align="left" style="background-color:#efefe4">(74) putamen_R</td>
<td valign="top" align="center" style="background-color:#efefe4">CO</td>
<td valign="top" align="center" style="background-color:#efefe4">0.74</td>
<td valign="top" align="center" style="background-color:#efefe4">0.68</td>
<td valign="top" align="center" style="background-color:#efefe4">0.06</td>
<td valign="top" align="center" style="background-color:#efefe4">0.27</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(30) Insula_R</td>
<td valign="top" align="center" style="background-color:#efefe4">CO</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">3</td>
<td valign="top" align="left" style="background-color:#efefe4">(31) Cingulum_Ant_L</td>
<td valign="top" align="center" style="background-color:#efefe4">CO</td>
<td valign="top" align="center" style="background-color:#efefe4">0.11</td>
<td valign="top" align="center" style="background-color:#efefe4">0.19</td>
<td valign="top" align="center" style="background-color:#efefe4">&#x02212;0.08</td>
<td valign="top" align="center" style="background-color:#efefe4">0.33</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(74) Putamen_R</td>
<td valign="top" align="center" style="background-color:#efefe4">CO</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#dd282d">4</td>
<td valign="top" align="left" style="background-color:#dd282d">(36) Cingulum_Post_R</td>
<td valign="top" align="center" style="background-color:#dd282d">CO</td>
<td valign="top" align="center" style="background-color:#dd282d">0.56</td>
<td valign="top" align="center" style="background-color:#dd282d">0.68</td>
<td valign="top" align="center" style="background-color:#dd282d">&#x02212;0.12</td>
<td valign="top" align="center" style="background-color:#dd282d">0.032</td>
</tr>
<tr>
<td style="background-color:#dd282d"/>
<td valign="top" align="left" style="background-color:#dd282d">(5) Frontal_Sup_Orb_L</td>
<td valign="top" align="center" style="background-color:#dd282d">CO</td>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">5</td>
<td valign="top" align="left" style="background-color:#efefe4">(39) ParaHippocampal_L</td>
<td valign="top" align="center" style="background-color:#efefe4">DM</td>
<td valign="top" align="center" style="background-color:#efefe4">0.89</td>
<td valign="top" align="center" style="background-color:#efefe4">0.83</td>
<td valign="top" align="center" style="background-color:#efefe4">0.06</td>
<td valign="top" align="center" style="background-color:#efefe4">0.45</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(90) Temporal_Inf_R</td>
<td valign="top" align="center" style="background-color:#efefe4">DM</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">6</td>
<td valign="top" align="left" style="background-color:#efefe4">(27) Rectus_L</td>
<td valign="top" align="center" style="background-color:#efefe4">DM</td>
<td valign="top" align="center" style="background-color:#efefe4">0.11</td>
<td valign="top" align="center" style="background-color:#efefe4">0.18</td>
<td valign="top" align="center" style="background-color:#efefe4">&#x02212;0.07</td>
<td valign="top" align="center" style="background-color:#efefe4">0.23</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(46) Cuneus_R</td>
<td valign="top" align="center" style="background-color:#efefe4">DM</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#28af57">7</td>
<td valign="top" align="left" style="background-color:#28af57">(89) Temporal_Inf_L</td>
<td valign="top" align="center" style="background-color:#28af57">DM</td>
<td valign="top" align="center" style="background-color:#28af57">0.88</td>
<td valign="top" align="center" style="background-color:#28af57">0.68</td>
<td valign="top" align="center" style="background-color:#28af57">0.2</td>
<td valign="top" align="center" style="background-color:#28af57">0.004</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(90) Parietal_Sup_L</td>
<td valign="top" align="center" style="background-color:#28af57">DM</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">8</td>
<td valign="top" align="left" style="background-color:#efefe4">(91) Cerebelum_Crus1_L</td>
<td valign="top" align="center" style="background-color:#efefe4">CB</td>
<td valign="top" align="center" style="background-color:#efefe4">0.04</td>
<td valign="top" align="center" style="background-color:#efefe4">0.11</td>
<td valign="top" align="center" style="background-color:#efefe4">&#x02212;0.07</td>
<td valign="top" align="center" style="background-color:#efefe4">0.12</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(100) Cerebelum_6_R</td>
<td valign="top" align="center" style="background-color:#efefe4">CB</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">9</td>
<td valign="top" align="left" style="background-color:#efefe4">(91) Cerebelum_Crus1_L</td>
<td valign="top" align="center" style="background-color:#efefe4">CB</td>
<td valign="top" align="center" style="background-color:#efefe4">0.76</td>
<td valign="top" align="center" style="background-color:#efefe4">0.83</td>
<td valign="top" align="center" style="background-color:#efefe4">&#x02212;0.07</td>
<td valign="top" align="center" style="background-color:#efefe4">0.17</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(108) Cerebelum _10_R</td>
<td valign="top" align="center" style="background-color:#efefe4">CB</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">10</td>
<td valign="top" align="left" style="background-color:#efefe4">(101) Creebelum_7b_L</td>
<td valign="top" align="center" style="background-color:#efefe4">CB</td>
<td valign="top" align="center" style="background-color:#efefe4">0.17</td>
<td valign="top" align="center" style="background-color:#efefe4">0.11</td>
<td valign="top" align="center" style="background-color:#efefe4">0.06</td>
<td valign="top" align="center" style="background-color:#efefe4">0.23</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(115) vermis_9</td>
<td valign="top" align="center" style="background-color:#efefe4">CB</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#28af57">11</td>
<td valign="top" align="left" style="background-color:#28af57">(32) Cingulum_Mid_L</td>
<td valign="top" align="center" style="background-color:#28af57">FP</td>
<td valign="top" align="center" style="background-color:#28af57">0.91</td>
<td valign="top" align="center" style="background-color:#28af57">0.74</td>
<td valign="top" align="center" style="background-color:#28af57">0.17</td>
<td valign="top" align="center" style="background-color:#28af57">0.0043</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(10) Frontal _ Inf_Orb_L</td>
<td valign="top" align="center" style="background-color:#28af57">FP</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">12</td>
<td valign="top" align="left" style="background-color:#efefe4">(10) Frontal _ Inf_Orb_L</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td valign="top" align="center" style="background-color:#efefe4">0.56</td>
<td valign="top" align="center" style="background-color:#efefe4">0.55</td>
<td valign="top" align="center" style="background-color:#efefe4">0.01</td>
<td valign="top" align="center" style="background-color:#efefe4">0.56</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(8) Frontal_Mid_R</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">13</td>
<td valign="top" align="left" style="background-color:#efefe4">(59) Parietal_Sup_L</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td valign="top" align="center" style="background-color:#efefe4">0.23</td>
<td valign="top" align="center" style="background-color:#efefe4">0.19</td>
<td valign="top" align="center" style="background-color:#efefe4">0.04</td>
<td valign="top" align="center" style="background-color:#efefe4">0.78</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(13) Frontal_Inf_Frontal_Tri_L</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#dd282d">14</td>
<td valign="top" align="left" style="background-color:#dd282d">(4) Frontal_Sup_R</td>
<td valign="top" align="center" style="background-color:#dd282d">CO</td>
<td valign="top" align="center" style="background-color:#dd282d">0.31</td>
<td valign="top" align="center" style="background-color:#dd282d">0.38</td>
<td valign="top" align="center" style="background-color:#dd282d">&#x02212;0.07</td>
<td valign="top" align="center" style="background-color:#dd282d">0.015</td>
</tr>
<tr>
<td style="background-color:#dd282d"/>
<td valign="top" align="left" style="background-color:#dd282d">(90) Temporal_Inf_R</td>
<td valign="top" align="center" style="background-color:#dd282d">DM</td>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#28af57">15</td>
<td valign="top" align="left" style="background-color:#28af57">(36) Cingulum_Post_R</td>
<td valign="top" align="center" style="background-color:#28af57">CO</td>
<td valign="top" align="center" style="background-color:#28af57">0.78</td>
<td valign="top" align="center" style="background-color:#28af57">0.66</td>
<td valign="top" align="center" style="background-color:#28af57">0.12</td>
<td valign="top" align="center" style="background-color:#28af57">0.002</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(90) Parietal_Sup_L</td>
<td valign="top" align="center" style="background-color:#28af57">DM</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">16</td>
<td valign="top" align="left" style="background-color:#efefe4">(32) Cingulum_Mid_L</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td valign="top" align="center" style="background-color:#efefe4">0.12</td>
<td valign="top" align="center" style="background-color:#efefe4">0.15</td>
<td valign="top" align="center" style="background-color:#efefe4">&#x02212;0.03</td>
<td valign="top" align="center" style="background-color:#efefe4">0.19</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(46) Cuneus_R</td>
<td valign="top" align="center" style="background-color:#efefe4">DM</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#dd282d">17</td>
<td valign="top" align="left" style="background-color:#dd282d">(30) Insula_R</td>
<td valign="top" align="center" style="background-color:#dd282d">CO</td>
<td valign="top" align="center" style="background-color:#dd282d">0.07</td>
<td valign="top" align="center" style="background-color:#dd282d">0.11</td>
<td valign="top" align="center" style="background-color:#dd282d">&#x02212;0.04</td>
<td valign="top" align="center" style="background-color:#dd282d">0.032</td>
</tr>
<tr>
<td style="background-color:#dd282d"/>
<td valign="top" align="left" style="background-color:#dd282d">(101) Cerebellum_7b_L</td>
<td valign="top" align="center" style="background-color:#dd282d">CB</td>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">18</td>
<td valign="top" align="left" style="background-color:#efefe4">(8) Frontal_Mid_R</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td valign="top" align="center" style="background-color:#efefe4">0.23</td>
<td valign="top" align="center" style="background-color:#efefe4">0.46</td>
<td valign="top" align="center" style="background-color:#efefe4">&#x02212;0.23</td>
<td valign="top" align="center" style="background-color:#efefe4">0.17</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(27) Rectus_L</td>
<td valign="top" align="center" style="background-color:#efefe4">DM</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#28af57">19</td>
<td valign="top" align="left" style="background-color:#28af57">(13) Frontal_Inf_Tri_L</td>
<td valign="top" align="center" style="background-color:#28af57">FP</td>
<td valign="top" align="center" style="background-color:#28af57">0.56</td>
<td valign="top" align="center" style="background-color:#28af57">0.44</td>
<td valign="top" align="center" style="background-color:#28af57">0.12</td>
<td valign="top" align="center" style="background-color:#28af57">0.026</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(74) Putamen_R</td>
<td valign="top" align="center" style="background-color:#28af57">CO</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">20</td>
<td valign="top" align="left" style="background-color:#efefe4">(89) Temporal_Inf_L</td>
<td valign="top" align="center" style="background-color:#efefe4">DM</td>
<td valign="top" align="center" style="background-color:#efefe4">0.68</td>
<td valign="top" align="center" style="background-color:#efefe4">0.65</td>
<td valign="top" align="center" style="background-color:#efefe4">0.03</td>
<td valign="top" align="center" style="background-color:#efefe4">0.34</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(36) Cingulum_Post_R</td>
<td valign="top" align="center" style="background-color:#efefe4">CO</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#28af57">21</td>
<td valign="top" align="left" style="background-color:#28af57">(91) Cerebelum_Crus1_L</td>
<td valign="top" align="center" style="background-color:#28af57">CB</td>
<td valign="top" align="center" style="background-color:#28af57">0.88</td>
<td valign="top" align="center" style="background-color:#28af57">0.75</td>
<td valign="top" align="center" style="background-color:#28af57">0.13</td>
<td valign="top" align="center" style="background-color:#28af57">0.045</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(90) Parietal_Sup_L</td>
<td valign="top" align="center" style="background-color:#28af57">DM</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">22</td>
<td valign="top" align="left" style="background-color:#efefe4">(110) Vermis_3</td>
<td valign="top" align="center" style="background-color:#efefe4">CB</td>
<td valign="top" align="center" style="background-color:#efefe4">0.23</td>
<td valign="top" align="center" style="background-color:#efefe4">0.18</td>
<td valign="top" align="center" style="background-color:#efefe4">0.05</td>
<td valign="top" align="center" style="background-color:#efefe4">0.34</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(13) Frontal_Inf_Tri_L</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#28af57">23</td>
<td valign="top" align="left" style="background-color:#28af57">(39) ParaHippocampal_L</td>
<td valign="top" align="center" style="background-color:#28af57">DM</td>
<td valign="top" align="center" style="background-color:#28af57">0.89</td>
<td valign="top" align="center" style="background-color:#28af57">0.84</td>
<td valign="top" align="center" style="background-color:#28af57">0.05</td>
<td valign="top" align="center" style="background-color:#28af57">0.036</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(30) Insula_R</td>
<td valign="top" align="center" style="background-color:#28af57">CO</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">24</td>
<td valign="top" align="left" style="background-color:#efefe4">(52) Occipital_Inf_L</td>
<td style="background-color:#efefe4"/>
<td valign="top" align="center" style="background-color:#efefe4">0.91</td>
<td valign="top" align="center" style="background-color:#efefe4">0.85</td>
<td valign="top" align="center" style="background-color:#efefe4">0.06</td>
<td valign="top" align="center" style="background-color:#efefe4">0.45</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(55) Fusiform_L</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">25</td>
<td valign="top" align="left" style="background-color:#efefe4">(14) Frontal_Inf_Tri_R</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td valign="top" align="center" style="background-color:#efefe4">0.23</td>
<td valign="top" align="center" style="background-color:#efefe4">0.16</td>
<td valign="top" align="center" style="background-color:#efefe4">0.07</td>
<td valign="top" align="center" style="background-color:#efefe4">0.76</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(42) Amygdala_L</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr> <tr>
<td valign="top" align="left" style="background-color:#efefe4">26</td>
<td valign="top" align="left" style="background-color:#efefe4">(71) Caudate_L</td>
<td style="background-color:#efefe4"/>
<td valign="top" align="center" style="background-color:#efefe4">0.35</td>
<td valign="top" align="center" style="background-color:#efefe4">0.32</td>
<td valign="top" align="center" style="background-color:#efefe4">0.03</td>
<td valign="top" align="center" style="background-color:#efefe4">0.68</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(14) Frontal_Inf_Tri_R</td>
<td valign="top" align="center" style="background-color:#efefe4">FP</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">27</td>
<td valign="top" align="left" style="background-color:#efefe4">(71) Caudate_L</td>
<td style="background-color:#efefe4"/>
<td valign="top" align="center" style="background-color:#efefe4">0.78</td>
<td valign="top" align="center" style="background-color:#efefe4">0.83</td>
<td valign="top" align="center" style="background-color:#efefe4">&#x02212;0.05</td>
<td valign="top" align="center" style="background-color:#efefe4">0.32</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(17) Rolandic_Oper_L</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">28</td>
<td valign="top" align="left" style="background-color:#efefe4">(71) Caudate_L</td>
<td style="background-color:#efefe4"/>
<td valign="top" align="center" style="background-color:#efefe4">0.87</td>
<td valign="top" align="center" style="background-color:#efefe4">0.84</td>
<td valign="top" align="center" style="background-color:#efefe4">0.09</td>
<td valign="top" align="center" style="background-color:#efefe4">0.61</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(25) Frontal_Med_Orb_L</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#efefe4">29</td>
<td valign="top" align="left" style="background-color:#efefe4">(81) Temporal_Sup_L</td>
<td valign="top" align="center" style="background-color:#efefe4">DM</td>
<td valign="top" align="center" style="background-color:#efefe4">0.12</td>
<td valign="top" align="center" style="background-color:#efefe4">0.18</td>
<td valign="top" align="center" style="background-color:#efefe4">&#x02212;0.06</td>
<td valign="top" align="center" style="background-color:#efefe4">0.12</td>
</tr>
<tr>
<td style="background-color:#efefe4"/>
<td valign="top" align="left" style="background-color:#efefe4">(90) Temporal_Inf_R</td>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
<td style="background-color:#efefe4"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#28af57">30</td>
<td valign="top" align="left" style="background-color:#28af57">(30) Insula_R</td>
<td valign="top" align="center" style="background-color:#28af57">CO</td>
<td valign="top" align="center" style="background-color:#28af57">0.12</td>
<td valign="top" align="center" style="background-color:#28af57">0.09</td>
<td valign="top" align="center" style="background-color:#28af57">0.03</td>
<td valign="top" align="center" style="background-color:#28af57">0.043</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(13) Frontal_Inf_Tri_L</td>
<td valign="top" align="center" style="background-color:#28af57">FP</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#dd282d">31</td>
<td valign="top" align="left" style="background-color:#dd282d">(31) Cingulum_Ant_L</td>
<td valign="top" align="center" style="background-color:#dd282d">CO</td>
<td valign="top" align="center" style="background-color:#dd282d">0.45</td>
<td valign="top" align="center" style="background-color:#dd282d">0.58</td>
<td valign="top" align="center" style="background-color:#dd282d">&#x02212;0.13</td>
<td valign="top" align="center" style="background-color:#dd282d">0.034</td>
</tr>
<tr>
<td style="background-color:#dd282d"/>
<td valign="top" align="left" style="background-color:#dd282d">(14) Frontal_Inf_Tri_R</td>
<td valign="top" align="center" style="background-color:#dd282d">FP</td>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
<td style="background-color:#dd282d"/>
</tr>
<tr>
<td valign="top" align="left" style="background-color:#28af57">32</td>
<td valign="top" align="left" style="background-color:#28af57">(30) Insula_R</td>
<td valign="top" align="center" style="background-color:#28af57">CO</td>
<td valign="top" align="center" style="background-color:#28af57">0.2</td>
<td valign="top" align="center" style="background-color:#28af57">0.23</td>
<td valign="top" align="center" style="background-color:#28af57">&#x02212;0.03</td>
<td valign="top" align="center" style="background-color:#28af57">0.036</td>
</tr>
<tr>
<td style="background-color:#28af57"/>
<td valign="top" align="left" style="background-color:#28af57">(90) Parietal_Sup_L</td>
<td valign="top" align="center" style="background-color:#28af57">DM</td>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
<td style="background-color:#28af57"/>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Red indicates the pair of regions are significantly negative correlated</italic>.</p>
<p><italic>Green indicates the pair of regions are significantly positive correlated</italic>.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>In the present study, a DNN model with a novel feature selection method was developed for classifying ASD patients and TD controls based on the whole-brain FCPs. The first contribution of this work is the proposed feature selection method based on multiple trained SAEs to improve the quality of low-dimension representations learned from stacked SAEs. The proposed method (DNN-FS) was able to select diverse features with high discriminating power from a large feature pool consisting of all features from multiple trained SAEs, and these selected features, in turn, led to better classification of ASD patients and TD controls compared to the performance of DNN-woFS, and models with other feature selection methods (two sample <italic>t</italic>-test and elastic net). To test the efficacy of the method, both DNN-FS and DNN-woFS systems were trained and evaluated by the five-fold nested CV scheme under different comparison scenarios. Both models had the same architecture (the same number of hidden layers/nodes) in each scenario, and different scenarios used different architectures. The DNN-FS (3/150 hidden layers/nodes) obtained the best classification accuracy of <bold>86.36%</bold>. Most importantly, the DNN-FS outperformed the DNN-woFS for each comparison scenario. The most significant improvement, of <bold>9.09%</bold>, occurred when the architecture contained three hidden layers with 150 nodes each. Among many ASD diagnosis models, Deshpande et al. (<xref ref-type="bibr" rid="B15">2013</xref>) developed a recursive cluster elimination based support vector machine classifier using effective connectivity weights, behavior assessment scores, FC, and fractional anisotropy obtained from DTI data. The model achieved a maximum classification accuracy of 95.9%. Unlike the work presented here, they combined analysis of multiple types of data. The feature selection part of the model was thus able to extract and integrate discriminating features from heterogeneous data, which potentially enhanced classification accuracy. In addition, the number of ROIs (18) defined in their model was small, which kept their input data dimension a reasonable size compared to the sample size (30 adolescents and young adults), and inhibited overfitting. However, this method still needs to be validated on a larger dataset. Nielsen et al. (<xref ref-type="bibr" rid="B53">2013</xref>) developed a classification model to perform ASD classification, but only achieved up to 60% accuracy. They collected data from multiple sites of ABIDE I, and different sites had different imaging protocols and quality control protocols. Training the model with a large dataset was the most likely reason for obtaining the low accuracy. In addition, the lower heterogeneity of the large dataset might be another reason for the low accuracy. Compared to the dataset (964) they used, the present study used a smaller dataset from one site of ABIDE I, so the data quality is expected to be higher. To avoid overfitting of the DNN model, a nested cross validation evaluation scheme was applied. The effectiveness of the proposed feature selection method was also compared with others feature selection methods: two sample <italic>t</italic>-test and elastic net. Results showed that the classification model trained by representations from the stacked SAEs outperformed the models trained by features from other two methods. A comparison of the average Fisher&#x00027;s score of features in each layer in the DNN-FS and DNN-woFS also showed that discriminative power increased layer-wise in both DNN-FS and DNN-woFS, with higher layers providing more discrimination.</p>
<p>The second contribution of this work is a DNN-based biomarker identification method to locate 32 FCs associated with ASD, and exploration of the biological implications of the findings. Among these 32 FC elements, 13 were within pre-defined brain networks including CO, DM, CB and FP, and 19 were between-network FCs. 13 FCs were statistically significant (<italic>p</italic> &#x0003C; 0.05) between ASD and TD groups. Five significant FCs (1, 17, 23, 30, and 32 in Table <xref ref-type="table" rid="T6">6</xref>) were associated with the insula. The previous study showed that this area was involved in interoceptive, affective, and empathic processes. Network analysis indicated that it was uniquely positioned as a hub mediating interactions between large-scale networks involved in externally- and internally-oriented cognitive processing, and it was a consistent locus of hypo- and hyper- activity in autism (Uddin and Menon, <xref ref-type="bibr" rid="B66">2009</xref>). Menon and Uddin (<xref ref-type="bibr" rid="B49">2010</xref>) have hypothesized that impaired hub function of the anterior insula would reduce the ability of people with ASD to flexibly move from the executive control networks to the default mode. In the findings, one FC (1 in Table <xref ref-type="table" rid="T6">6</xref>) was within-network (CO), and four FCs were between-network (one CO and CB, two DM and CO, one CO and CP). The results indicated that both hyper- and hypo-connectivities associated with the insula were existed within certain network or between networks, which can back up the point view in literature. Four significant FCs (4, 11, 15, and 31 in Table <xref ref-type="table" rid="T6">6</xref>) were detected associated with the cingulate cortex. The right posterior cingulum were strongly correlated with the superior parietal gyrus, and were weakly correlated with the superior frontal gyrus in the ASD group. The left anterior cingulum was weakly correlated with the right pars triangularis, and the left middle cingulum was strongly correlated with the left pars orbitalis in the ASD group. The connections of the cingulate cortex to other brain structures are extensive, and thus the functions of the region are varied and complex. It makes important contributions to emotion, and various types of cognition such as the decision-making and the management of social behavior (Ikuta et al., <xref ref-type="bibr" rid="B32">2014</xref>). The findings can help people to explore the neurological basis associated with abnormal symptoms in emotion and cognition. Meanwhile, another four ASD-related significant FCs were identified: LITG was strongly connected to LSPL (7 in Table <xref ref-type="table" rid="T6">6</xref>), RSFG was weakly connected with RITG (14 in Table <xref ref-type="table" rid="T6">6</xref>), LPT was strongly connected RP (19 in Table <xref ref-type="table" rid="T6">6</xref>), and LCCB was strongly connected to LSPL (21 in Table <xref ref-type="table" rid="T6">6</xref>). Brain areas associated with these FCs are involved in different brain functions such as spatial orientation (LSPL), visual shape processing (LITG&#x00026;RITG), self-awareness (RSFG), and motor skills (LPT; Frackowiak, <xref ref-type="bibr" rid="B21">2004</xref>; Lee et al., <xref ref-type="bibr" rid="B41">2007</xref>). The dysfunctionality of certain brain areas might lead to such abnormal FCs in autistic children. However, the mechanism is unclear, and needed to be explored in the future. Other than above findings, 19 FCs which were not statistical significant were detected by the method too. They are not, respectively, discriminating may be because the sample size was not large enough to build the significant statistical power for each of them. However, they should be important in the classification task because all FCs made contribution during the learning process, and the model performance is supposed to be decreased if any of them is missing.</p>
<p>In the future, we are going to extend the current research in two aspects. First of all, the proposed method will be tested on datasets from cohorts in different age groups. Age is an important factor in the ASD diagnosis. ASD biomarkers may be altered due to the age difference. It is especially meaningful to make a convincing prediction before children develop ASD symptoms at their early age. Hazlett et al. (<xref ref-type="bibr" rid="B26">2017</xref>) demonstrated that they can make reasonably accurate forecast about which of these high-risk infants will later develop ASD themselves by examining the growth rate of brain volume of infants between 6 and 24 months based on anatomical brain images. The finding motivates us to evaluate the proposed method on a dataset from early-age children. NDAR (Hall et al., <xref ref-type="bibr" rid="B24">2012</xref>) is an NIH-funded research data repository that aims to accelerate progress in ASD research, and it contains neuroimaging datasets from cohorts in different age groups. We plan to validate the generalization of the method on multiple NDAR datasets from infants, adolescents, and adults. We are optimistic about the model classification performance. No matter what the age group the dataset belongs to, the strength difference of certain FCs should exist between ASD patients and the age-matched TD controls. The DNN-FS can not only capture the discriminating FCs but also can learn high-quality discriminating features from the whole-brain FCP to enhance the classification accuracy of the SR model. However, the identified ASD-related FCs may be altered in different age groups. So it is necessary to work with neuroscientists closely to explore the neurological basis of identified FCs in each age groups.</p>
<p>Second of all, more systematic methods will be developed to decide the size of the feature pool for the proposed feature selection method, and to determine how the size affects the final performance. In addition, we will continue to explore the way of deciding the architecture (No. of layers/No. of nodes per layer) of the DNN-FS. It was found that the performance of the DNN-FS did not increase by adding more hidden layer nodes when the number of hidden layers was fixed. The most plausible explanation is that additional hidden layer nodes may learn many redundant features, which decreases the feature learning capacity of the stacked SAEs and further reduces classification accuracy. It was also found that the performance of the DNN-FS cannot be increased simply by adding more hidden layers when the number of hidden layer nodes is fixed. The most reasonable explanation is that the gradients vanish quickly in lower layers when the DNN has a very deep architecture, and even the pre-training step is not able to help much. Both these issues will be studied further as part of future research to obtain optimal DNN-FSs for classifying ASD patients and TD controls.</p>
</sec>
<sec id="s5">
<title>Ethics statement</title>
<p>The Autism data used in this research was acquired through the public ABIDE I database through a data use agreement. The Autism database has de-identified all the patient health information (PHI) associated with the data. The original study to collect patients data was approved by IRB at University of Michigan. The potential risks of this project are limited to the unauthorized usage of the Autism data. To mitigate the potential risk, we will take the utmost care to ensure that there is no abuse of protected information.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>XG, LL, and AM conceived the project idea. XG implemented the method and performed the experiments. LL supervised the project. AM supervised the development of the feature selection method. KD and CE provided the detailed analysis of brain FCs discovered by the method. HL provided critical suggestions for the experiments design.</p>
<sec>
<title>Conflict of interest statement</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
</sec>
</body>
<back>
<ack><p>The project is supported by a methodology grant to LL as part of an Institutional Clinical and Translational Science Award, NIH/NCATS Grant Number 8UL1TR000077-05, National Natural Science Foundation of China No. 31601083 and the National Institutes of Health (NIH) Clinical and Translational Science Award (CTSA) program, grant 5UL1TR001425-03. HL is partially supported by an NIH grant R01HL111829.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alaerts</surname> <given-names>K.</given-names></name> <name><surname>Nayar</surname> <given-names>K.</given-names></name> <name><surname>Kelly</surname> <given-names>C.</given-names></name> <name><surname>Raithel</surname> <given-names>J.</given-names></name> <name><surname>Milham</surname> <given-names>M. P.</given-names></name> <name><surname>Di Martino</surname> <given-names>A.</given-names></name></person-group> (<year>2015</year>). <article-title>Age-related changes in intrinsic function of the superior temporal sulcus in autism spectrum disorders</article-title>. <source>Soc. Cogn. Affect. Neurosci.</source> <volume>10</volume>, <fpage>1413</fpage>&#x02013;<lpage>1423</lpage>. <pub-id pub-id-type="doi">10.1093/scan/nsv029</pub-id><pub-id pub-id-type="pmid">25809403</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="book"><person-group person-group-type="author"><collab>American Psychiatric Association</collab></person-group> (<year>2013</year>). <source>Diagnostic and Statistical Manual of Mental Disorders (DSM-5&#x000AE;)</source>. <publisher-loc>Arlington, VA</publisher-loc>: <publisher-name>American Psychiatric Publishing</publisher-name>.</citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Assaf</surname> <given-names>M.</given-names></name> <name><surname>Jagannathan</surname> <given-names>K.</given-names></name> <name><surname>Calhoun</surname> <given-names>V. D.</given-names></name> <name><surname>Miller</surname> <given-names>L.</given-names></name> <name><surname>Stevens</surname> <given-names>M. C.</given-names></name> <name><surname>Sahl</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Abnormal functional connectivity of default mode sub-networks in autism spectrum disorder patients</article-title>. <source>Neuroimage</source> <volume>53</volume>, <fpage>247</fpage>&#x02013;<lpage>256</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2010.05.067</pub-id><pub-id pub-id-type="pmid">20621638</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Lamblin</surname> <given-names>P.</given-names></name> <name><surname>Popovici</surname> <given-names>D.</given-names></name> <name><surname>Larochelle</surname> <given-names>H.</given-names></name></person-group> (<year>2007</year>). <article-title>Greedy layer-wise training of deep networks</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>19</volume>:<fpage>153</fpage>.</citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Biswal</surname> <given-names>B.</given-names></name> <name><surname>Zerrin Yetkin</surname> <given-names>F.</given-names></name> <name><surname>Haughton</surname> <given-names>V. M.</given-names></name> <name><surname>Hyde</surname> <given-names>J. S.</given-names></name></person-group> (<year>1995</year>). <article-title>Functional connectivity in the motor cortex of resting human brain using echo-planar mri</article-title>. <source>Magn. Reson. Med.</source> <volume>34</volume>, <fpage>537</fpage>&#x02013;<lpage>541</lpage>. <pub-id pub-id-type="doi">10.1002/mrm.1910340409</pub-id><pub-id pub-id-type="pmid">8524021</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bos</surname> <given-names>D. J.</given-names></name> <name><surname>van Raalten</surname> <given-names>T. R.</given-names></name> <name><surname>Oranje</surname> <given-names>B.</given-names></name> <name><surname>Smits</surname> <given-names>A. R.</given-names></name> <name><surname>Kobussen</surname> <given-names>N. A.</given-names></name> <name><surname>van Belle</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Developmental differences in higher-order resting-state networks in Autism Spectrum Disorder</article-title>. <source>Neuroimage</source> <volume>4</volume>, <fpage>820</fpage>&#x02013;<lpage>827</lpage>. <pub-id pub-id-type="doi">10.1016/j.nicl.2014.05.007</pub-id><pub-id pub-id-type="pmid">24936432</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Brosch</surname> <given-names>T.</given-names></name> <name><surname>Tam</surname> <given-names>R.</given-names></name> <collab>for the Alzheimer&#x00027;s Disease Neuroimaging Initiative</collab></person-group> (<year>2013</year>). <article-title>Manifold learning of brain MRIs by deep learning</article-title>, in <source>International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Berlin; Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>633</fpage>&#x02013;<lpage>640</lpage>.</citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cerliani</surname> <given-names>L.</given-names></name> <name><surname>Mennes</surname> <given-names>M.</given-names></name> <name><surname>Thomas</surname> <given-names>R. M.</given-names></name> <name><surname>Di Martino</surname> <given-names>A.</given-names></name> <name><surname>Thioux</surname> <given-names>M.</given-names></name> <name><surname>Keysers</surname> <given-names>C.</given-names></name></person-group> (<year>2015</year>). <article-title>Increased functional connectivity between subcortical and cortical resting-state networks in autism spectrum disorder</article-title>. <source>JAMA Psychiatry</source> <volume>72</volume>, <fpage>767</fpage>&#x02013;<lpage>777</lpage>. <pub-id pub-id-type="doi">10.1001/jamapsychiatry.2015.0101</pub-id><pub-id pub-id-type="pmid">26061743</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chase</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>Alzheimer disease: altered functional connectivity in preclinical dementia</article-title>. <source>Nat. Rev. Neurol.</source> <volume>10</volume>, <fpage>609</fpage>&#x02013;<lpage>609</lpage>. <pub-id pub-id-type="doi">10.1038/nrneurol.2014.195</pub-id><pub-id pub-id-type="pmid">25330722</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>C. P.</given-names></name> <name><surname>Keown</surname> <given-names>C. L.</given-names></name> <name><surname>Jahedi</surname> <given-names>A.</given-names></name> <name><surname>Nair</surname> <given-names>A.</given-names></name> <name><surname>Pflieger</surname> <given-names>M. E.</given-names></name> <name><surname>Bailey</surname> <given-names>B. A.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Diagnostic classification of intrinsic functional connectivity highlights somatosensory, default mode, and visual regions in autism</article-title>. <source>Neuroimage</source> <volume>8</volume>, <fpage>238</fpage>&#x02013;<lpage>245</lpage>. <pub-id pub-id-type="doi">10.1016/j.nicl.2015.04.002</pub-id><pub-id pub-id-type="pmid">26106547</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Kang</surname> <given-names>J.</given-names></name> <name><surname>Wang</surname> <given-names>G.</given-names></name></person-group> (<year>2015</year>). <article-title>An empirical Bayes normalization method for connectivity metrics in resting state fMRI</article-title>. <source>Front. Neurosci.</source> <volume>9</volume>:<fpage>316</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2015.00316</pub-id><pub-id pub-id-type="pmid">26441493</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cherkassky</surname> <given-names>V. L.</given-names></name> <name><surname>Kana</surname> <given-names>R. K.</given-names></name> <name><surname>Keller</surname> <given-names>T. A.</given-names></name> <name><surname>Just</surname> <given-names>M. A.</given-names></name></person-group> (<year>2006</year>). <article-title>Functional connectivity in a baseline resting-state network in autism</article-title>. <source>Neuroreport</source> <volume>17</volume>, <fpage>1687</fpage>&#x02013;<lpage>1690</lpage>. <pub-id pub-id-type="doi">10.1097/01.wnr.0000239956.45448.4c</pub-id><pub-id pub-id-type="pmid">17047454</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Darken</surname> <given-names>C.</given-names></name> <name><surname>Moody</surname> <given-names>J. E.</given-names></name></person-group> (<year>1990</year>). <source>Note on Learning Rate Schedules for Stochastic Optimization</source>. <publisher-loc>Denver</publisher-loc>: <publisher-name>NIPS</publisher-name>.</citation></ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Denil</surname> <given-names>M.</given-names></name> <name><surname>Shakibi</surname> <given-names>B.</given-names></name> <name><surname>Dinh</surname> <given-names>L.</given-names></name> <name><surname>de Freitas</surname> <given-names>N.</given-names></name></person-group> (<year>2013</year>). <article-title>Predicting parameters in deep learning</article-title>, in <source>Advances in Neural Information Processing Systems</source>, eds <person-group person-group-type="editor"><name><surname>Burges</surname> <given-names>C. J. C.</given-names></name> <name><surname>Bottou</surname> <given-names>L.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name> <name><surname>Ghahramani</surname> <given-names>Z.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<publisher-loc>South Lake Tahoe, NV</publisher-loc>: <publisher-name>Neural Information Processing Systems Foundation, Inc.</publisher-name>), <fpage>2148</fpage>&#x02013;<lpage>2156</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://papers.nips.cc/book/advances-in-neural-information-processing-systems-26-2013">https://papers.nips.cc/book/advances-in-neural-information-processing-systems-26-2013</ext-link></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deshpande</surname> <given-names>G.</given-names></name> <name><surname>Libero</surname> <given-names>L. E.</given-names></name> <name><surname>Sreenivasan</surname> <given-names>K. R.</given-names></name> <name><surname>Deshpande</surname> <given-names>H. D.</given-names></name> <name><surname>Kana</surname> <given-names>R. K.</given-names></name></person-group> (<year>2013</year>). <article-title>Identification of neural connectivity signatures of autism using machine learning</article-title>. <source>Front. Hum. Neurosci.</source> <volume>7</volume>:<fpage>670</fpage>. <pub-id pub-id-type="doi">10.3389/fnhum.2013.00670</pub-id><pub-id pub-id-type="pmid">24151458</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Di Martino</surname> <given-names>A.</given-names></name> <name><surname>Yan</surname> <given-names>C.-G.</given-names></name> <name><surname>Li</surname> <given-names>Q.</given-names></name> <name><surname>Denio</surname> <given-names>E.</given-names></name> <name><surname>Castellanos</surname> <given-names>F. X.</given-names></name> <name><surname>Alaerts</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>The autism brain imaging data exchange: towards a large-scale evaluation of the intrinsic brain architecture in autism</article-title>. <source>Mol. Psychiatry</source> <volume>19</volume>, <fpage>659</fpage>&#x02013;<lpage>667</lpage>. <pub-id pub-id-type="doi">10.1038/mp.2013.78</pub-id><pub-id pub-id-type="pmid">23774715</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Doyle-Thomas</surname> <given-names>K. A.</given-names></name> <name><surname>Lee</surname> <given-names>W.</given-names></name> <name><surname>Foster</surname> <given-names>N. E.</given-names></name> <name><surname>Tryfon</surname> <given-names>A.</given-names></name> <name><surname>Ouimet</surname> <given-names>T.</given-names></name> <name><surname>Hyde</surname> <given-names>K. L.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Atypical functional brain connectivity during rest in autism spectrum disorders</article-title>. <source>Ann. Neurol.</source> <volume>77</volume>, <fpage>866</fpage>&#x02013;<lpage>876</lpage>. <pub-id pub-id-type="doi">10.1002/ana.24391</pub-id><pub-id pub-id-type="pmid">25707715</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dunn</surname> <given-names>L. M.</given-names></name> <name><surname>Dunn</surname> <given-names>L. M.</given-names></name> <name><surname>Bulheller</surname> <given-names>S.</given-names></name> <name><surname>H&#x000E4;cker</surname> <given-names>H.</given-names></name></person-group> (<year>1965</year>). <source>Peabody Picture Vocabulary Test</source>. <publisher-loc>Circle Pines, MN</publisher-loc>: <publisher-name>American Guidance Service</publisher-name>.</citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Erhan</surname> <given-names>D.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Courville</surname> <given-names>A.</given-names></name> <name><surname>Manzagol</surname> <given-names>P.-A.</given-names></name> <name><surname>Vincent</surname> <given-names>P.</given-names></name> <name><surname>Bengio</surname> <given-names>S.</given-names></name></person-group> (<year>2010</year>). <article-title>Why does unsupervised pre-training help deep learning?</article-title> <source>J. Machine Learn. Res.</source> <volume>11</volume>, <fpage>625</fpage>&#x02013;<lpage>660</lpage>.</citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fair</surname> <given-names>D. A.</given-names></name> <name><surname>Nigg</surname> <given-names>J. T.</given-names></name> <name><surname>Iyer</surname> <given-names>S.</given-names></name> <name><surname>Bathula</surname> <given-names>D.</given-names></name> <name><surname>Mills</surname> <given-names>K. L.</given-names></name> <name><surname>Dosenbach</surname> <given-names>N. U.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Distinct neural signatures detected for ADHD subtypes after controlling for micro-movements in resting state functional connectivity MRI data</article-title>. <source>Front. Syst. Neurosci.</source> <volume>6</volume>:<fpage>80</fpage>. <pub-id pub-id-type="doi">10.3389/fnsys.2012.00080</pub-id><pub-id pub-id-type="pmid">23382713</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Frackowiak</surname> <given-names>R. S.</given-names></name></person-group> (<year>2004</year>). <source>Human Brain Function</source>. <publisher-loc>Cambridge, MA</publisher-loc>: <publisher-name>Academic press</publisher-name>.</citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Glover</surname> <given-names>G. H.</given-names></name> <name><surname>Law</surname> <given-names>C. S.</given-names></name></person-group> (<year>2001</year>). <article-title>Spiral-in/out BOLD fMRI for increased SNR and reduced susceptibility artifacts</article-title>. <source>Magn. Reson. Med.</source> <volume>46</volume>, <fpage>515</fpage>&#x02013;<lpage>522</lpage>. <pub-id pub-id-type="doi">10.1002/mrm.1222</pub-id><pub-id pub-id-type="pmid">11550244</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Graves</surname> <given-names>A.</given-names></name> <name><surname>Mohamed</surname> <given-names>A.-R.</given-names></name> <name><surname>Hinton</surname> <given-names>G.</given-names></name></person-group> (<year>2013</year>). <article-title>Speech recognition with deep recurrent neural networks</article-title>, in <source>2013 IEEE International Conference on Acoustics</source> (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>Speech and Signal Processing; IEEE</publisher-name>), <fpage>6645</fpage>&#x02013;<lpage>6649</lpage>.</citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hall</surname> <given-names>D.</given-names></name> <name><surname>Huerta</surname> <given-names>M. F.</given-names></name> <name><surname>McAuliffe</surname> <given-names>M. J.</given-names></name> <name><surname>Farber</surname> <given-names>G. K.</given-names></name></person-group> (<year>2012</year>). <article-title>Sharing heterogeneous data: the national database for autism research</article-title>. <source>Neuroinformatics</source> <volume>10</volume>, <fpage>331</fpage>&#x02013;<lpage>339</lpage>. <pub-id pub-id-type="doi">10.1007/s12021-012-9151-4</pub-id><pub-id pub-id-type="pmid">22622767</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Haykin</surname> <given-names>S. S.</given-names></name> <name><surname>Kosko</surname> <given-names>B.</given-names></name></person-group> (<year>2001</year>). <source>Intelligent Signal Processing</source>. <publisher-loc>Hoboken, NJ</publisher-loc>: <publisher-name>Wiley-IEEE Press</publisher-name>.</citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hazlett</surname> <given-names>H. C.</given-names></name> <name><surname>Gu</surname> <given-names>H.</given-names></name> <name><surname>Munsell</surname> <given-names>B. C.</given-names></name> <name><surname>Kim</surname> <given-names>S. H.</given-names></name> <name><surname>Styner</surname> <given-names>M.</given-names></name> <name><surname>Wolff</surname> <given-names>J. J.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Early brain development in infants at high risk for autism spectrum disorder</article-title>. <source>Nature</source> <volume>542</volume>, <fpage>348</fpage>&#x02013;<lpage>351</lpage>. <pub-id pub-id-type="doi">10.1038/nature21369</pub-id><pub-id pub-id-type="pmid">28202961</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2007</year>). <article-title>To recognize shapes, first learn to generate images</article-title>. <source>Prog. Brain Res.</source> <volume>165</volume>, <fpage>535</fpage>&#x02013;<lpage>547</lpage>. <pub-id pub-id-type="doi">10.1016/S0079-6123(06)65034-6</pub-id><pub-id pub-id-type="pmid">17925269</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hinton</surname> <given-names>G. E.</given-names></name> <name><surname>Osindero</surname> <given-names>S.</given-names></name> <name><surname>Teh</surname> <given-names>Y.-W.</given-names></name></person-group> (<year>2006</year>). <article-title>A fast learning algorithm for deep belief nets</article-title>. <source>Neural Comput.</source> <volume>18</volume>, <fpage>1527</fpage>&#x02013;<lpage>1554</lpage>. <pub-id pub-id-type="doi">10.1162/neco.2006.18.7.1527</pub-id><pub-id pub-id-type="pmid">16764513</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hjelm</surname> <given-names>R. D.</given-names></name> <name><surname>Calhoun</surname> <given-names>V. D.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name> <name><surname>Allen</surname> <given-names>E. A.</given-names></name> <name><surname>Adali</surname> <given-names>T.</given-names></name> <name><surname>Plis</surname> <given-names>S. M.</given-names></name></person-group> (<year>2014</year>). <article-title>Restricted Boltzmann machines for neuroimaging: an application in identifying intrinsic networks</article-title>. <source>NeuroImage</source> <volume>96</volume>, <fpage>245</fpage>&#x02013;<lpage>260</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2014.03.048</pub-id><pub-id pub-id-type="pmid">24680869</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hong</surname> <given-names>C.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name> <name><surname>Wan</surname> <given-names>J.</given-names></name> <name><surname>Tao</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>M.</given-names></name></person-group> (<year>2015</year>). <article-title>Multimodal deep autoencoder for human pose recovery</article-title>. <source>IEEE Trans. Image Process.</source> <volume>24</volume>, <fpage>5659</fpage>&#x02013;<lpage>5670</lpage>. <pub-id pub-id-type="doi">10.1109/TIP.2015.2487860</pub-id><pub-id pub-id-type="pmid">26452284</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Iidaka</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). <article-title>Resting state functional magnetic resonance imaging and neural network classified autism and control</article-title>. <source>Cortex</source> <volume>63</volume>, <fpage>55</fpage>&#x02013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1016/j.cortex.2014.08.011</pub-id><pub-id pub-id-type="pmid">25243989</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ikuta</surname> <given-names>T.</given-names></name> <name><surname>Shafritz</surname> <given-names>K. M.</given-names></name> <name><surname>Bregman</surname> <given-names>J.</given-names></name> <name><surname>Peters</surname> <given-names>B. D.</given-names></name> <name><surname>Gruner</surname> <given-names>P.</given-names></name> <name><surname>Malhotra</surname> <given-names>A. K.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Abnormal cingulum bundle development in autism: a probabilistic tractography study</article-title>. <source>Psychiatry Res.</source> <volume>221</volume>, <fpage>63</fpage>&#x02013;<lpage>68</lpage>. <pub-id pub-id-type="doi">10.1016/j.pscychresns.2013.08.002</pub-id><pub-id pub-id-type="pmid">24231056</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Itahashi</surname> <given-names>T.</given-names></name> <name><surname>Yamada</surname> <given-names>T.</given-names></name> <name><surname>Watanabe</surname> <given-names>H.</given-names></name> <name><surname>Nakamura</surname> <given-names>M.</given-names></name> <name><surname>Jimbo</surname> <given-names>D.</given-names></name> <name><surname>Shioda</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Altered network topologies and hub organization in adults with autism: a resting-state fMRI study</article-title>. <source>PLoS ONE</source> <volume>9</volume>:<fpage>e94115</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0094115</pub-id><pub-id pub-id-type="pmid">24714805</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Itahashi</surname> <given-names>T.</given-names></name> <name><surname>Yamada</surname> <given-names>T.</given-names></name> <name><surname>Watanabe</surname> <given-names>H.</given-names></name> <name><surname>Nakamura</surname> <given-names>M.</given-names></name> <name><surname>Ohta</surname> <given-names>H.</given-names></name> <name><surname>Kanai</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Alterations of local spontaneous brain activity and connectivity in adults with high-functioning autism spectrum disorder</article-title>. <source>Mol. Autism</source> <volume>6</volume>:<fpage>30</fpage>. <pub-id pub-id-type="doi">10.1186/s13229-015-0026-z</pub-id><pub-id pub-id-type="pmid">26023326</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jann</surname> <given-names>K.</given-names></name> <name><surname>Hernandez</surname> <given-names>L. M.</given-names></name> <name><surname>Beck-Pancer</surname> <given-names>D.</given-names></name> <name><surname>McCarron</surname> <given-names>R.</given-names></name> <name><surname>Smith</surname> <given-names>R. X.</given-names></name> <name><surname>Dapretto</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Altered resting perfusion and functional connectivity of default mode network in youth with autism spectrum disorder</article-title>. <source>Brain Behav.</source> <volume>5</volume>:<fpage>e00358</fpage>. <pub-id pub-id-type="doi">10.1002/brb3.358</pub-id><pub-id pub-id-type="pmid">26445698</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jung</surname> <given-names>M.</given-names></name> <name><surname>Mody</surname> <given-names>M.</given-names></name> <name><surname>Saito</surname> <given-names>D. N.</given-names></name> <name><surname>Tomoda</surname> <given-names>A.</given-names></name> <name><surname>Okazawa</surname> <given-names>H.</given-names></name> <name><surname>Wada</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Sex differences in the default mode network with regard to autism spectrum traits: a resting state fMRI study</article-title>. <source>PloS ONE</source> <volume>10</volume>:<fpage>e0143126</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0143126</pub-id><pub-id pub-id-type="pmid">26600385</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Keown</surname> <given-names>C. L.</given-names></name> <name><surname>Shih</surname> <given-names>P.</given-names></name> <name><surname>Nair</surname> <given-names>A.</given-names></name> <name><surname>Peterson</surname> <given-names>N.</given-names></name> <name><surname>Mulvey</surname> <given-names>M. E.</given-names></name> <name><surname>M&#x000FC;ller</surname> <given-names>R.-A.</given-names></name></person-group> (<year>2013</year>). <article-title>Local functional overconnectivity in posterior brain regions is associated with symptom severity in autism spectrum disorders</article-title>. <source>Cell Rep.</source> <volume>5</volume>, <fpage>567</fpage>&#x02013;<lpage>572</lpage>. <pub-id pub-id-type="doi">10.1016/j.celrep.2013.10.003</pub-id><pub-id pub-id-type="pmid">24210815</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>J.</given-names></name> <name><surname>Calhoun</surname> <given-names>V. D.</given-names></name> <name><surname>Shim</surname> <given-names>E.</given-names></name> <name><surname>Lee</surname> <given-names>J.-H.</given-names></name></person-group> (<year>2016</year>). <article-title>Deep neural network with weight sparsity control and pre-training extracts hierarchical features and enhances classification performance: evidence from whole-brain resting-state functional connectivity patterns of schizophrenia</article-title>. <source>Neuroimage</source> <volume>124</volume>, <fpage>127</fpage>&#x02013;<lpage>146</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2015.05.018</pub-id><pub-id pub-id-type="pmid">25987366</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2012</year>). <article-title>Imagenet classification with deep convolutional neural networks</article-title>, in <source>Advances in Neural Information Processing Systems</source>, eds <person-group person-group-type="editor"><name><surname>Pereira</surname> <given-names>F.</given-names></name> <name><surname>Burges</surname> <given-names>C. J. C.</given-names></name> <name><surname>Bottou</surname> <given-names>L.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<publisher-loc>South Lake Tahoe, NV</publisher-loc>: <publisher-name>Neural Information Processing Systems Foundation, Inc.</publisher-name>), <fpage>1097</fpage>&#x02013;<lpage>1105</lpage>.</citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Larochelle</surname> <given-names>H.</given-names></name> <name><surname>Bengio</surname> <given-names>Y.</given-names></name> <name><surname>Louradour</surname> <given-names>J.</given-names></name> <name><surname>Lamblin</surname> <given-names>P.</given-names></name></person-group> (<year>2009</year>). <article-title>Exploring strategies for training deep neural networks</article-title>. <source>J. Machine Learn. Res.</source> <volume>10</volume>, <fpage>1</fpage>&#x02013;<lpage>40</lpage>.</citation></ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>H.</given-names></name> <name><surname>Ekanadham</surname> <given-names>C.</given-names></name> <name><surname>Ng</surname> <given-names>A. Y.</given-names></name></person-group> (<year>2007</year>). <article-title>Sparse deep belief net model for visual area V2</article-title>, in <source>Advances in Neural Information Processing Systems</source>, eds <person-group person-group-type="editor"><name><surname>Platt</surname> <given-names>J. C.</given-names></name> <name><surname>Koller</surname> <given-names>D.</given-names></name> <name><surname>Singer</surname> <given-names>Y.</given-names></name> <name><surname>Roweis</surname> <given-names>S. T.</given-names></name></person-group> (<publisher-loc>Vancouver, BC</publisher-loc>: <publisher-name>Neural Information Processing Systems Foundation, Inc.</publisher-name>), <fpage>873</fpage>&#x02013;<lpage>880</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://papers.nips.cc/book/advances-in-neural-information-processing-systems-20-2007">https://papers.nips.cc/book/advances-in-neural-information-processing-systems-20-2007</ext-link></citation></ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>D. C.</given-names></name> <name><surname>Nocedal</surname> <given-names>J.</given-names></name></person-group> (<year>1989</year>). <article-title>On the limited memory BFGS method for large scale optimization</article-title>. <source>Math Program</source> <volume>45</volume>, <fpage>503</fpage>&#x02013;<lpage>528</lpage>. <pub-id pub-id-type="doi">10.1007/BF01589116</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lord</surname> <given-names>C.</given-names></name> <name><surname>Risi</surname> <given-names>S.</given-names></name> <name><surname>DiLavore</surname> <given-names>P. S.</given-names></name> <name><surname>Shulman</surname> <given-names>C.</given-names></name> <name><surname>Thurm</surname> <given-names>A.</given-names></name> <name><surname>Pickles</surname> <given-names>A.</given-names></name></person-group> (<year>2006</year>). <article-title>Autism from 2 to 9 years of age</article-title>. <source>Arch. Gen. Psychiatry</source> <volume>63</volume>, <fpage>694</fpage>&#x02013;<lpage>701</lpage>. <pub-id pub-id-type="doi">10.1001/archpsyc.63.6.694</pub-id><pub-id pub-id-type="pmid">16754843</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lord</surname> <given-names>C.</given-names></name> <name><surname>Risi</surname> <given-names>S.</given-names></name> <name><surname>Lambrecht</surname> <given-names>L.</given-names></name> <name><surname>Cook</surname> <given-names>E. H.</given-names> <suffix>Jr.</suffix></name> <name><surname>Leventhal</surname> <given-names>B. L.</given-names></name> <name><surname>DiLavore</surname> <given-names>P. C.</given-names></name> <etal/></person-group>. (<year>2000</year>). <article-title>The Autism Diagnostic Observation Schedule&#x02014;Generic: a standard measure of social and communication deficits associated with the spectrum of autism</article-title>. <source>J. Autism Dev. Disord.</source> <volume>30</volume>, <fpage>205</fpage>&#x02013;<lpage>223</lpage>. <pub-id pub-id-type="doi">10.1023/A:1005592401947</pub-id><pub-id pub-id-type="pmid">11055457</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lord</surname> <given-names>C.</given-names></name> <name><surname>Rutter</surname> <given-names>M.</given-names></name> <name><surname>Le Couteur</surname> <given-names>A.</given-names></name></person-group> (<year>1994</year>). <article-title>Autism Diagnostic Interview-Revised: a revised version of a diagnostic interview for caregivers of individuals with possible pervasive developmental disorders</article-title>. <source>J. Autism Dev. Disord.</source> <volume>24</volume>, <fpage>659</fpage>&#x02013;<lpage>685</lpage>. <pub-id pub-id-type="doi">10.1007/BF02172145</pub-id><pub-id pub-id-type="pmid">7814313</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lynall</surname> <given-names>M.-E.</given-names></name> <name><surname>Bassett</surname> <given-names>D. S.</given-names></name> <name><surname>Kerwin</surname> <given-names>R.</given-names></name> <name><surname>McKenna</surname> <given-names>P. J.</given-names></name> <name><surname>Kitzbichler</surname> <given-names>M.</given-names></name> <name><surname>Muller</surname> <given-names>U.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Functional connectivity and brain networks in schizophrenia</article-title>. <source>J. Neurosci.</source> <volume>30</volume>, <fpage>9477</fpage>&#x02013;<lpage>9487</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.0333-10.2010</pub-id><pub-id pub-id-type="pmid">20631176</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mandell</surname> <given-names>D. S.</given-names></name> <name><surname>Ittenbach</surname> <given-names>R. F.</given-names></name> <name><surname>Levy</surname> <given-names>S. E.</given-names></name> <name><surname>Pinto-Martin</surname> <given-names>J. A.</given-names></name></person-group> (<year>2007</year>). <article-title>Disparities in diagnoses received prior to a diagnosis of autism spectrum disorder</article-title>. <source>J. Autism Dev. Disord.</source> <volume>37</volume>, <fpage>1795</fpage>&#x02013;<lpage>1802</lpage>. <pub-id pub-id-type="doi">10.1007/s10803-006-0314-8</pub-id><pub-id pub-id-type="pmid">17160456</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Martin</surname> <given-names>A. R.</given-names></name> <name><surname>Aleksanderek</surname> <given-names>I.</given-names></name> <name><surname>Cohen-Adad</surname> <given-names>J.</given-names></name> <name><surname>Tarmohamed</surname> <given-names>Z.</given-names></name> <name><surname>Tetreault</surname> <given-names>L.</given-names></name> <name><surname>Smith</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Translating state-of-the-art spinal cord MRI techniques to clinical use: a systematic review of clinical studies utilizing DTI, MT, MWF, MRS, and fMRI</article-title>. <source>Neuroimage</source> <volume>10</volume>, <fpage>192</fpage>&#x02013;<lpage>238</lpage>. <pub-id pub-id-type="doi">10.1016/j.nicl.2015.11.019</pub-id><pub-id pub-id-type="pmid">26862478</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Menon</surname> <given-names>V.</given-names></name> <name><surname>Uddin</surname> <given-names>L. Q.</given-names></name></person-group> (<year>2010</year>). <article-title>Saliency, switching, attention and control: a network model of insula function</article-title>. <source>Brain Struct. Funct.</source> <volume>214</volume>, <fpage>655</fpage>&#x02013;<lpage>667</lpage>. <pub-id pub-id-type="doi">10.1007/s00429-010-0262-0</pub-id><pub-id pub-id-type="pmid">20512370</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Monk</surname> <given-names>C. S.</given-names></name> <name><surname>Peltier</surname> <given-names>S. J.</given-names></name> <name><surname>Wiggins</surname> <given-names>J. L.</given-names></name> <name><surname>Weng</surname> <given-names>S.-J.</given-names></name> <name><surname>Carrasco</surname> <given-names>M.</given-names></name> <name><surname>Risi</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Abnormalities of intrinsic functional connectivity in autism spectrum disorders</article-title>. <source>Neuroimage</source> <volume>47</volume>, <fpage>764</fpage>&#x02013;<lpage>772</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2009.04.069</pub-id><pub-id pub-id-type="pmid">19409498</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moody</surname> <given-names>J.</given-names></name> <name><surname>Hanson</surname> <given-names>S.</given-names></name> <name><surname>Krogh</surname> <given-names>A.</given-names></name> <name><surname>Hertz</surname> <given-names>J. A.</given-names></name></person-group> (<year>1995</year>). <article-title>A simple weight decay can improve generalization</article-title>. <source>Adv. Neural Inform. Process. Syst.</source> <volume>4</volume>, <fpage>950</fpage>&#x02013;<lpage>957</lpage>.</citation></ref>
<ref id="B52">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ngiam</surname> <given-names>J.</given-names></name> <name><surname>Khosla</surname> <given-names>A.</given-names></name> <name><surname>Kim</surname> <given-names>M.</given-names></name> <name><surname>Nam</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>H.</given-names></name> <name><surname>Ng</surname> <given-names>A. Y.</given-names></name></person-group> (<year>2011</year>). <article-title>Multimodal deep learning</article-title>, in <source>Proceedings of the 28th International Conference on Machine Learning (ICML-11)</source> (<publisher-loc>Bellevue</publisher-loc>), <fpage>689</fpage>&#x02013;<lpage>696</lpage>.</citation></ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nielsen</surname> <given-names>J. A.</given-names></name> <name><surname>Zielinski</surname> <given-names>B. A.</given-names></name> <name><surname>Fletcher</surname> <given-names>P. T.</given-names></name> <name><surname>Alexander</surname> <given-names>A. L.</given-names></name> <name><surname>Lange</surname> <given-names>N.</given-names></name> <name><surname>Bigler</surname> <given-names>E. D.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Multisite functional connectivity MRI classification of autism: ABIDE results</article-title>. <source>Front. Hum. Neurosci.</source> <volume>7</volume>:<fpage>599</fpage>. <pub-id pub-id-type="doi">10.3389/fnhum.2013.00599</pub-id><pub-id pub-id-type="pmid">24093016</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nylander</surname> <given-names>L.</given-names></name> <name><surname>Holmqvist</surname> <given-names>M.</given-names></name> <name><surname>Gustafson</surname> <given-names>L.</given-names></name> <name><surname>Gillberg</surname> <given-names>C.</given-names></name></person-group> (<year>2013</year>). <article-title>Attention-deficit/hyperactivity disorder (ADHD) and autism spectrum disorder (ASD) in adult psychiatry. A 20-year register study</article-title>. <source>Nord. J. Psychiatry</source> <volume>67</volume>, <fpage>344</fpage>&#x02013;<lpage>350</lpage>. <pub-id pub-id-type="doi">10.3109/08039488.2012.748824</pub-id><pub-id pub-id-type="pmid">23234539</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Plis</surname> <given-names>S. M.</given-names></name> <name><surname>Hjelm</surname> <given-names>D. R.</given-names></name> <name><surname>Salakhutdinov</surname> <given-names>R.</given-names></name> <name><surname>Calhoun</surname> <given-names>V. D.</given-names></name></person-group> (<year>2013</year>). <article-title>Deep learning for neuroimaging: a validation study</article-title>. arXiv preprint arXiv:1312.5847.</citation></ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Power</surname> <given-names>J. D.</given-names></name> <name><surname>Barnes</surname> <given-names>K. A.</given-names></name> <name><surname>Snyder</surname> <given-names>A. Z.</given-names></name> <name><surname>Schlaggar</surname> <given-names>B. L.</given-names></name> <name><surname>Petersen</surname> <given-names>S. E.</given-names></name></person-group> (<year>2012</year>). <article-title>Spurious but systematic correlations in functional connectivity MRI networks arise from subject motion</article-title>. <source>Neuroimage</source> <volume>59</volume>, <fpage>2142</fpage>&#x02013;<lpage>2154</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2011.10.018</pub-id><pub-id pub-id-type="pmid">22019881</pub-id></citation></ref>
<ref id="B57">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Rosner</surname> <given-names>B.</given-names></name></person-group> (<year>2015</year>). <source>Fundamentals of Biostatistics.</source> <publisher-loc>Toronto, ON</publisher-loc>: <publisher-name>Nelson Education</publisher-name>.</citation></ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Satterthwaite</surname> <given-names>T. D.</given-names></name> <name><surname>Elliott</surname> <given-names>M. A.</given-names></name> <name><surname>Gerraty</surname> <given-names>R. T.</given-names></name> <name><surname>Ruparel</surname> <given-names>K.</given-names></name> <name><surname>Loughead</surname> <given-names>J.</given-names></name> <name><surname>Calkins</surname> <given-names>M. E.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>An improved framework for confound regression and filtering for control of motion artifact in the preprocessing of resting-state functional connectivity data</article-title>. <source>Neuroimage</source> <volume>64</volume>, <fpage>240</fpage>&#x02013;<lpage>256</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2012.08.052</pub-id><pub-id pub-id-type="pmid">22926292</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shin</surname> <given-names>H.-C.</given-names></name> <name><surname>Orton</surname> <given-names>M. R.</given-names></name> <name><surname>Collins</surname> <given-names>D. J.</given-names></name> <name><surname>Doran</surname> <given-names>S. J.</given-names></name> <name><surname>Leach</surname> <given-names>M. O.</given-names></name></person-group> (<year>2013</year>). <article-title>Stacked autoencoders for unsupervised feature learning and multiple organ detection in a pilot study using 4D patient data</article-title>. <source>Pattern Anal. Mach. Intell. IEEE Trans.</source> <volume>35</volume>, <fpage>1930</fpage>&#x02013;<lpage>1943</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2012.277</pub-id><pub-id pub-id-type="pmid">23787345</pub-id></citation></ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Starck</surname> <given-names>T.</given-names></name> <name><surname>Nikkinen</surname> <given-names>J.</given-names></name> <name><surname>Rahko</surname> <given-names>J.</given-names></name> <name><surname>Remes</surname> <given-names>J.</given-names></name> <name><surname>Hurtig</surname> <given-names>T.</given-names></name> <name><surname>Haapsamo</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Resting state fMRI reveals a default mode dissociation between retrosplenial and medial prefrontal subnetworks in ASD despite motion scrubbing</article-title>. <source>Front. Hum. Neurosci.</source> <volume>7</volume>:<fpage>802</fpage>. <pub-id pub-id-type="doi">10.3389/fnhum.2013.00802</pub-id><pub-id pub-id-type="pmid">24319422</pub-id></citation></ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Suk</surname> <given-names>H.-I.</given-names></name> <name><surname>Lee</surname> <given-names>S.-W.</given-names></name> <name><surname>Shen</surname> <given-names>D.</given-names></name> <collab>for the Alzheimer&#x00027;s Disease Neuroimaging Initiative</collab></person-group> (<year>2014</year>). <article-title>Hierarchical feature representation and multimodal fusion with deep learning for AD/MCI diagnosis</article-title>. <source>Neuroimage</source> <volume>101</volume>, <fpage>569</fpage>&#x02013;<lpage>582</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2014.06.077</pub-id><pub-id pub-id-type="pmid">25042445</pub-id></citation></ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Suk</surname> <given-names>H.-I.</given-names></name> <name><surname>Lee</surname> <given-names>S.-W.</given-names></name> <name><surname>Shen</surname> <given-names>D.</given-names></name> <collab>for the Alzheimer&#x00027;s Disease Neuroimaging Initiative</collab></person-group> (<year>2015</year>). <article-title>Latent feature representation with stacked auto-encoder for AD/MCI diagnosis</article-title>. <source>Brain Struct. Funct.</source> <volume>220</volume>, <fpage>841</fpage>&#x02013;<lpage>859</lpage>. <pub-id pub-id-type="doi">10.1007/s00429-013-0687-3</pub-id><pub-id pub-id-type="pmid">24363140</pub-id></citation></ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Supekar</surname> <given-names>K.</given-names></name> <name><surname>Uddin</surname> <given-names>L. Q.</given-names></name> <name><surname>Khouzam</surname> <given-names>A.</given-names></name> <name><surname>Phillips</surname> <given-names>J.</given-names></name> <name><surname>Gaillard</surname> <given-names>W. D.</given-names></name> <name><surname>Kenworthy</surname> <given-names>L. E.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Brain hyperconnectivity in children with autism and its links to social deficits</article-title>. <source>Cell Rep.</source> <volume>5</volume>, <fpage>738</fpage>&#x02013;<lpage>747</lpage>. <pub-id pub-id-type="doi">10.1016/j.celrep.2013.10.001</pub-id><pub-id pub-id-type="pmid">24210821</pub-id></citation></ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tyszka</surname> <given-names>J. M.</given-names></name> <name><surname>Kennedy</surname> <given-names>D. P.</given-names></name> <name><surname>Paul</surname> <given-names>L. K.</given-names></name> <name><surname>Adolphs</surname> <given-names>R.</given-names></name></person-group> (<year>2013</year>). <article-title>Largely typical patterns of resting-state functional connectivity in high-functioning adults with autism</article-title>. <source>Cereb. Cortex</source> <volume>24</volume>, <fpage>1894</fpage>&#x02013;<lpage>1905</lpage>. <pub-id pub-id-type="doi">10.1093/cercor/bht040</pub-id><pub-id pub-id-type="pmid">23425893</pub-id></citation></ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tzourio-Mazoyer</surname> <given-names>N.</given-names></name> <name><surname>Landeau</surname> <given-names>B.</given-names></name> <name><surname>Papathanassiou</surname> <given-names>D.</given-names></name> <name><surname>Crivello</surname> <given-names>F.</given-names></name> <name><surname>Etard</surname> <given-names>O.</given-names></name> <name><surname>Delcroix</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2002</year>). <article-title>Automated anatomical labeling of activations in SPM using a macroscopic anatomical parcellation of the MNI MRI single-subject brain</article-title>. <source>Neuroimage</source> <volume>15</volume>, <fpage>273</fpage>&#x02013;<lpage>289</lpage>. <pub-id pub-id-type="doi">10.1006/nimg.2001.0978</pub-id><pub-id pub-id-type="pmid">11771995</pub-id></citation></ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Uddin</surname> <given-names>L. Q.</given-names></name> <name><surname>Menon</surname> <given-names>V.</given-names></name></person-group> (<year>2009</year>). <article-title>The anterior insula in autism: under-connected and under-examined</article-title>. <source>Neurosci. Biobehav. Rev.</source> <volume>33</volume>, <fpage>1198</fpage>&#x02013;<lpage>1203</lpage>. <pub-id pub-id-type="doi">10.1016/j.neubiorev.2009.06.002</pub-id><pub-id pub-id-type="pmid">19538989</pub-id></citation></ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Uddin</surname> <given-names>L. Q.</given-names></name> <name><surname>Supekar</surname> <given-names>K.</given-names></name> <name><surname>Lynch</surname> <given-names>C. J.</given-names></name> <name><surname>Khouzam</surname> <given-names>A.</given-names></name> <name><surname>Phillips</surname> <given-names>J.</given-names></name> <name><surname>Feinstein</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Salience network&#x02013;based classification and prediction of symptom severity in children with autism</article-title>. <source>JAMA Psychiatry</source> <volume>70</volume>, <fpage>869</fpage>&#x02013;<lpage>879</lpage>. <pub-id pub-id-type="doi">10.1001/jamapsychiatry.2013.104</pub-id><pub-id pub-id-type="pmid">23803651</pub-id></citation></ref>
<ref id="B68">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Werbos</surname> <given-names>P.</given-names></name></person-group> (<year>1974</year>). <source>Beyond Regression: New Tools for Prediction and Analysis in the Behavioral Sciences</source>. Doctoral Dissertation, <publisher-name>Applied Mathematics, Harvard University</publisher-name>, <publisher-loc>Cambridge, MA</publisher-loc>.</citation></ref>
<ref id="B69">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Weston</surname> <given-names>J.</given-names></name> <name><surname>Mukherjee</surname> <given-names>S.</given-names></name> <name><surname>Chapelle</surname> <given-names>O.</given-names></name> <name><surname>Pontil</surname> <given-names>M.</given-names></name> <name><surname>Poggio</surname> <given-names>T.</given-names></name> <name><surname>Vapnik</surname> <given-names>V.</given-names></name></person-group> (<year>2000</year>). <article-title>Feature selection for SVMs</article-title>, in <source>Advances in Neural Information Processing Systems</source> (<publisher-loc>Denver, CO</publisher-loc>: <publisher-name>Neural Information Processing Systems Foundation, Inc.</publisher-name>), <fpage>668</fpage>&#x02013;<lpage>674</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://papers.nips.cc/book/advances-in-neural-information-processing-systems-13-2000">https://papers.nips.cc/book/advances-in-neural-information-processing-systems-13-2000</ext-link></citation></ref>
<ref id="B70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wiggins</surname> <given-names>J. L.</given-names></name> <name><surname>Peltier</surname> <given-names>S. J.</given-names></name> <name><surname>Ashinoff</surname> <given-names>S.</given-names></name> <name><surname>Weng</surname> <given-names>S.-J.</given-names></name> <name><surname>Carrasco</surname> <given-names>M.</given-names></name> <name><surname>Welsh</surname> <given-names>R. C.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Using a self-organizing map algorithm to detect age-related changes in functional connectivity during rest in autism spectrum disorders</article-title>. <source>Brain Res.</source> <volume>1380</volume>, <fpage>187</fpage>&#x02013;<lpage>197</lpage>. <pub-id pub-id-type="doi">10.1016/j.brainres.2010.10.102</pub-id><pub-id pub-id-type="pmid">21047495</pub-id></citation></ref>
<ref id="B71">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yahata</surname> <given-names>N.</given-names></name> <name><surname>Morimoto</surname> <given-names>J.</given-names></name> <name><surname>Hashimoto</surname> <given-names>R.</given-names></name> <name><surname>Lisi</surname> <given-names>G.</given-names></name> <name><surname>Shibata</surname> <given-names>K.</given-names></name> <name><surname>Kawakubo</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>A small number of abnormal brain connections predicts adult autism spectrum disorder</article-title>. <source>Nat. Commun.</source> <volume>7</volume>:<fpage>11254</fpage>. <pub-id pub-id-type="doi">10.1038/ncomms11254</pub-id><pub-id pub-id-type="pmid">27075704</pub-id></citation></ref>
<ref id="B72">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yan</surname> <given-names>C.-G.</given-names></name> <name><surname>Cheung</surname> <given-names>B.</given-names></name> <name><surname>Kelly</surname> <given-names>C.</given-names></name> <name><surname>Colcombe</surname> <given-names>S.</given-names></name> <name><surname>Craddock</surname> <given-names>R. C.</given-names></name> <name><surname>Di Martino</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>A comprehensive assessment of regional variation in the impact of head micromovements on functional connectomics</article-title>. <source>Neuroimage</source> <volume>76</volume>, <fpage>183</fpage>&#x02013;<lpage>201</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2013.03.004</pub-id><pub-id pub-id-type="pmid">23499792</pub-id></citation></ref>
<ref id="B73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>L.</given-names></name> <name><surname>Hu</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name></person-group> (<year>2015</year>). <article-title>Heterogeneous feature selection with multi-modal deep neural networks and sparse group lasso</article-title>. <source>IEEE Trans. Multimedia</source> <volume>17</volume>, <fpage>1936</fpage>&#x02013;<lpage>1948</lpage>. <pub-id pub-id-type="doi">10.1109/TMM.2015.2477058</pub-id></citation></ref>
<ref id="B74">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>Z.</given-names></name> <name><surname>Luo</surname> <given-names>P.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Tang</surname> <given-names>X.</given-names></name></person-group> (<year>2014</year>). <article-title>Deep learning multi-view representation for face recognition</article-title>. arXiv preprint arXiv:1406.6947.</citation></ref>
<ref id="B75">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zou</surname> <given-names>H.</given-names></name> <name><surname>Hastie</surname> <given-names>T.</given-names></name></person-group> (<year>2005</year>). <article-title>Regularization and variable selection via the elastic net</article-title>. <source>J. R. Stat. Soc. Ser. B</source> <volume>67</volume>, <fpage>301</fpage>&#x02013;<lpage>320</lpage>. <pub-id pub-id-type="doi">10.1111/j.1467-9868.2005.00503.x</pub-id></citation></ref>
</ref-list>
</back>
</article>
