<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Psychiatry</journal-id>
<journal-title>Frontiers in Psychiatry</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Psychiatry</abbrev-journal-title>
<issn pub-type="epub">1664-0640</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpsyt.2025.1643552</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Psychiatry</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Predicting alcohol use disorder risk in firefighters using a multimodal deep learning model: a cross-sectional study</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Jang</surname>
<given-names>MyeongGyun</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3089399/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
</contrib>
<contrib contrib-type="author" equal-contrib="yes">
<name>
<surname>Kim</surname>
<given-names>DongOk</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3037703/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Yoon</surname>
<given-names>Sujung</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/423807/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Lee</surname>
<given-names>Hwamin</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2887224/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Department of Biomedical Informatics, Korea University College of Medicine</institution>, <addr-line>Seoul</addr-line>,&#xa0;<country>Republic of Korea</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Ewha Brain Institute, Ewha Womans University</institution>, <addr-line>Seoul</addr-line>,&#xa0;<country>Republic of Korea</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Department of Brain and Cognitive Sciences, Ewha Womans University</institution>, <addr-line>Seoul</addr-line>,&#xa0;<country>Republic of Korea</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/1527597/overview">Ulrich Wesemann</ext-link>, Military Hospital Berlin, Germany</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/805557/overview">Filippo Rapisarda</ext-link>, Universit&#xe9; du Qu&#xe9;bec &#xe0; Trois-Rivi&#xe8;res, Canada</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3173025/overview">Sumit Shinde</ext-link>, Pune Institute of Computer Technology, India</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Sujung Yoon, <email xlink:href="mailto:sujungjyoon@ewha.ac.kr">sujungjyoon@ewha.ac.kr</email>; Hwamin Lee, <email xlink:href="mailto:hwamin@korea.ac.kr">hwamin@korea.ac.kr</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>03</day>
<month>11</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>16</volume>
<elocation-id>1643552</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>02</day>
<month>10</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Jang, Kim, Yoon and Lee.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Jang, Kim, Yoon and Lee</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Firefighters constitute a high-risk occupational cohort for alcohol use disorder (AUD) due to chronic trauma exposure, yet traditional screening methodologies relying on self-report instruments remain compromised by systematic underreporting attributable to occupational stigma and career preservation concerns. This cross-sectional investigation developed and validated a multimodal deep learning framework integrating T1-weighted structural magnetic resonance imaging with standardized neuropsychological assessments to enable objective AUD risk stratification without necessitating computationally intensive functional neuroimaging protocols.</p>
</sec>
<sec>
<title>Methods</title>
<p>Analysis of 689 active-duty firefighters (mean age 43.3&#xb1;8.8 years; 93% male) from a nationwide occupational cohort incorporated high-resolution three-dimensional T1-weighted structural MRI acquisition alongside comprehensive neuropsychological evaluation utilizing the Grooved Pegboard Test for visual-motor coordination assessment and Trail Making Test for executive function quantification. The novel computational architecture synergistically combined ResNet-50 convolutional neural networks for hierarchical morphological feature extraction, Vision Transformer modules for global neuroanatomical pattern recognition, and multilayer perceptron integration of clinical variables, with model interpretability assessed through Gradient-weighted Class Activation Mapping and SHapley Additive exPlanations methodologies. Performance evaluation employed stratified three-fold cross-validation with DeLong's test for statistical comparison of receiver operating characteristic curves.</p>
</sec>
<sec>
<title>Results</title>
<p>The multimodal framework achieved 79.88% classification accuracy with area under the receiver operating characteristic curve of 79.65%, representing statistically significant performance enhancement relative to clinical-only (62.53%; p&lt;0.001) and neuroimaging-only (61.53%; p&lt;0.001) models, demonstrating a 17.35 percentage-point improvement attributable to synergistic cross-modal integration rather than simple feature concatenation. Interpretability analyses revealed stochastic activation patterns in unimodal neuroimaging models lacking neuroanatomically coherent feature localization, while clinical feature importance hierarchically prioritized biological sex and motor coordination metrics as primary predictive indicators. The framework maintained robust calibration across probability thresholds, supporting operational feasibility for clinical deployment.</p>
</sec>
<sec>
<title>Discussion</title>
<p>This investigation establishes that structural neuroimaging combined with targeted neuropsychological assessment achieves classification performance comparable to complex multimodal protocols while substantially reducing acquisition time and computational requirements, offering a pragmatic pathway for implementing objective AUD screening in high-risk occupational populations with broader implications for psychiatric risk stratification in trauma-exposed professions.</p>
</sec>
</abstract>
<kwd-group>
<kwd>alcohol use disorder</kwd>
<kwd>firefighters</kwd>
<kwd>multimodal deep learning</kwd>
<kwd>structural MRI</kwd>
<kwd>occupational psychiatry</kwd>
<kwd>neuroimaging biomarkers</kwd>
</kwd-group>
<contract-num rid="cn001">RS-2024-00457381, RS-2024-00440371</contract-num>
<contract-sponsor id="cn001">National Research Foundation of Korea<named-content content-type="fundref-id">10.13039/501100003725</named-content>
</contract-sponsor>
<counts>
<fig-count count="4"/>
<table-count count="3"/>
<equation-count count="2"/>
<ref-count count="68"/>
<page-count count="18"/>
<word-count count="9716"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Neuroimaging</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Firefighters constitute a distinct occupational group regularly exposed to life-threatening emergencies and cumulative psychological trauma including fire suppression, technical rescues, hazardous material responses, and mass casualty incidents. This continuous exposure imposes substantial psychological and physiological burdens, placing firefighters at elevated risk for a range of mental health disorders, most notably alcohol use disorder (AUD) (<xref ref-type="bibr" rid="B1">1</xref>). Epidemiological studies have consistently reported higher rates of problematic alcohol consumption among firefighters compared to the general population, a disparity that persists even after adjusting for demographic and socioeconomic factors (<xref ref-type="bibr" rid="B2">2</xref>&#x2013;<xref ref-type="bibr" rid="B4">4</xref>).</p>
<p>Beyond alcohol-specific outcomes, large-scale evidence from Canadian public safety personnel (PSP) shows substantially elevated screening rates for common mental disorders relative to the general population. In a national survey of 5,813 PSP, Carleton et&#xa0;al. (<xref ref-type="bibr" rid="B5">5</xref>) reported that 15.1% screened positive for at least one current disorder and 26.7% for two or more, with meaningful differences across PSP categories (<xref ref-type="bibr" rid="B5">5</xref>). These findings underscore the high and heterogeneous mental health burden in firefighters&#x2019; broader occupational context and help explain why coping-motivated alcohol use often emerges in this workforce, reinforcing the need for objective, stigma-resistant risk assessment beyond self-report. This pattern is consistent with evidence that public safety personnel, including firefighters, frequently engage in coping-motivated alcohol use in response to trauma and chronic operational stress (<xref ref-type="bibr" rid="B6">6</xref>&#x2013;<xref ref-type="bibr" rid="B9">9</xref>), further strengthening the rationale for objective risk assessment methods. The etiology of AUD within this population is multifaceted, reflecting interactions among neurobiological predispositions, occupational stress, and psychosocial dynamics. Alcohol is often utilized as a maladaptive coping strategy to manage symptoms of hyperarousal, intrusive memories, and emotional distress stemming from repeated trauma exposure (<xref ref-type="bibr" rid="B7">7</xref>&#x2013;<xref ref-type="bibr" rid="B9">9</xref>). Over time, this reliance on alcohol for emotional regulation can lead to reinforcement cycles that escalate into habitual and dependent use (<xref ref-type="bibr" rid="B10">10</xref>). These clinical risk pathways are further compounded by occupational culture. Firefighting environments frequently normalize post-shift drinking and valorize stoicism, creating a paradox in which alcohol use is both institutionally sanctioned and individually stigmatized (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B10">10</xref>). Consequently, many firefighters refrain from help-seeking behaviors and underreport their alcohol consumption due to fears of career-related repercussions.</p>
<p>The implications of AUD within firefighting populations extend beyond individual health, impacting operational readiness, decision-making under pressure, and public safety during emergency response. Excessive alcohol use among first responders in high-stakes environments is linked to increased risk-taking behaviors, such as driving while intoxicated, thereby contributing to significant occupational problems that can affect team performance, and posing severe threats to personal safety, including heightened suicidality and increased risk of traumatic incidents (<xref ref-type="bibr" rid="B11">11</xref>). Despite these risks, early detection of alcohol misuse remains challenging. Current screening protocols rely heavily on self-reported questionnaires such as the Alcohol Use Disorder Identification Test (AUDIT), which are vulnerable to social desirability bias, impression management, and concerns regarding occupational repercussions (<xref ref-type="bibr" rid="B12">12</xref>). Furthermore, cultural norms emphasizing resilience and self-reliance may suppress disclosure of substance use and deter engagement with support services (<xref ref-type="bibr" rid="B13">13</xref>). Accordingly, there is a clear need for objective, stigma-resistant screening approaches that integrate biological and behavioral indicators rather than relying solely on self-report.</p>
<p>Recent advances in neuroimaging and machine learning have opened new avenues for objective assessment of psychiatric disorders. Structural MRI markers have been shown to correlate with various psychiatric phenotypes, including those related to substance use disorders (<xref ref-type="bibr" rid="B14">14</xref>). Machine learning techniques applied to neuroimaging data have demonstrated promising diagnostic and predictive accuracy across various psychiatric disorders (<xref ref-type="bibr" rid="B15">15</xref>, <xref ref-type="bibr" rid="B16">16</xref>). However, their application to alcohol use risk prediction within occupational cohorts remains underexplored. Within high-risk occupational cohorts such as firefighters, studies that objectively predict AUD risk by integrating structural MRI with standardized neuropsychological measures remain scarce.</p>
<p>To address this gap, the present study proposes a multimodal deep learning approach that integrates neuroimaging features with clinical and cognitive measures to predict AUD risk in a national sample of active-duty firefighters. This method aims to overcome limitations of conventional self-report tools by leveraging biologically informed, data-driven markers to enhance early identification of high-risk individuals. Through this integration, we seek to contribute to the development of precision screening strategies tailored to the unique demands and vulnerabilities of high-stress emergency response professionals.</p>
</sec>
<sec id="s2" sec-type="materials|methods">
<label>2</label>
<title>Materials and methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Study design and participants</title>
<p>This study utilized a cross-sectional design to develop and evaluate a multimodal deep learning framework for predicting alcohol use disorder (AUD) risk in an occupational cohort of active-duty firefighters in the Republic of Korea. Participants were recruited from multiple fire stations nationwide. Eligibility criteria included: age 25&#x2013;65 years, active employment as a firefighter, and availability of both T1-weighted structural magnetic resonance imaging (MRI) and complete clinical assessment data. Exclusion criteria comprised a history of neurological disorders (e.g., epilepsy, stroke, traumatic brain injury), major psychiatric conditions other than AUD, current use of psychotropic medications, MRI-detected structural brain abnormalities, or contraindications to MRI scanning (e.g., metallic implants, claustrophobia).</p>
<p>Of 746 initially enrolled firefighters, 35 were excluded due to incomplete imaging data, 14 for missing clinical assessments, and 8 for MRI-detected structural anomalies, resulting in a final analytical sample of 689 participants (mean age 43.3 &#xb1; 8.8 years; 93% male). <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> illustrates the participant recruitment and data preprocessing workflow. All participants provided written informed consent, and the study protocol was approved by the Institutional Review Board of Ewha Womans University. The research adhered to the ethical principles of the Declaration of Helsinki.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Participant recruitment and data preprocessing workflow for the firefighter cohort.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1643552-g001.tif">
<alt-text content-type="machine-generated">Flowchart depicting participant selection for a study. Initially, 746 participants were recruited, with 26 missing alcohol risk values. Of the remaining 720, 21 had missing grooved pegboard test values. This left 699 participants, from whom 10 had missing trail making test values, resulting in 689 final participants. These were divided into a non-alcohol risk group of 297 and an alcohol risk group of 392.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Clinical assessments</title>
<p>Cognitive and motor functions were evaluated using two standardized neuropsychological tests: the Grooved Pegboard Test and the Trail Making Test (TMT) (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B18">18</xref>) (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B20">20</xref>). The Grooved Pegboard Test assessed visual-motor coordination and fine motor control. Participants inserted 25 uniquely shaped pins into corresponding grooves as quickly as possible, with completion times (seconds) recorded for both dominant and non-dominant hands; longer times indicated poorer performance (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B18">18</xref>). The TMT evaluated processing speed, cognitive flexibility, and executive function. Part A required participants to connect numbered circles sequentially, while Part B involved alternating between numbers and letters in ascending order. Completion times were recorded, with higher values reflecting lower cognitive efficiency (<xref ref-type="bibr" rid="B19">19</xref>, <xref ref-type="bibr" rid="B20">20</xref>). Tests were administered by trained personnel under standardized conditions.</p>
<p>Alcohol use risk was assessed using the Alcohol Use Disorder Identification Test (AUDIT), a 10-item self-report questionnaire developed by the World Health Organization to evaluate alcohol consumption, dependence symptoms, and related harm (<xref ref-type="bibr" rid="B21">21</xref>). Scores range from 0 to 40, with a cutoff of &#x2265;8 indicating hazardous drinking risk (<xref ref-type="bibr" rid="B22">22</xref>). The AUDIT was completed under supervised conditions to ensure data integrity.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Demographic and neuropsychological characteristics</title>
<p>To align comparisons with standard occupational screening practice, we stratified the cohort using the established AUDIT cut-off (&#x2265;8 vs&lt;8). <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> summarizes the demographic and neuropsychological characteristics of the study cohort, stratified by AUDIT-based alcohol risk status (&#x2265;8: alcohol risk, n=392, 56.9%;&lt;8: non-alcohol risk, n=297, 43.1%). The alcohol risk group had a mean age of 43.16 &#xb1; 8.53 years, compared to 42.58 &#xb1; 8.67 years for the non-alcohol risk group, with no significant difference (p=0.380, two-tailed independent samples t-test). We used two-tailed independent-samples t-tests for continuous variables because the groups are non-overlapping at the participant level, the t-test provides an efficient test of mean differences, and with our sample size it is reasonably robust to moderate deviations from normality; a two-sided test also guards against effects in either direction. A significant gender disparity was observed (p&lt;0.001, two-tailed Pearson Chi-square test), with the alcohol risk group showing higher male predominance (380 males, 12 females) compared to the non-alcohol risk group (257 males, 40 females). The Pearson chi-square test was chosen for categorical comparisons (e.g., sex distribution) because it assesses association between group membership and categorical outcomes without requiring distributional assumptions beyond adequate expected cell counts.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Participant characteristics stratified by alcohol use risk status.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Variables</th>
<th valign="middle" align="left">Alcohol risk</th>
<th valign="middle" align="left">Non-alcohol risk</th>
<th valign="middle" align="left">p value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Sample size</td>
<td valign="middle" align="left">392</td>
<td valign="middle" align="left">297</td>
<td valign="middle" align="left"/>
</tr>
<tr>
<td valign="middle" align="left">Age (years) <xref ref-type="table-fn" rid="fnT1_1">
<sup>a</sup>
</xref>
</td>
<td valign="middle" align="left">43.16 &#xb1; 8.53</td>
<td valign="middle" align="left">42.58 &#xb1; 8.67</td>
<td valign="middle" align="left">0.380 <xref ref-type="table-fn" rid="fnT1_2">
<sup>b</sup>
</xref>
</td>
</tr>
<tr>
<td valign="middle" align="left">Gender (male/female)</td>
<td valign="middle" align="left">380/12</td>
<td valign="middle" align="left">257/40</td>
<td valign="middle" align="left">&lt; 0.001 <xref ref-type="table-fn" rid="fnT1_3">
<sup>c</sup>
</xref>
</td>
</tr>
<tr>
<th valign="middle" colspan="4" align="left">Grooved pegboard test</th>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;Dominant <xref ref-type="table-fn" rid="fnT1_1">
<sup>a</sup>
</xref>
</td>
<td valign="middle" align="left">66.27 &#xb1; 8.45</td>
<td valign="middle" align="left">67.27 &#xb1; 9.46</td>
<td valign="middle" align="left">0.153 <xref ref-type="table-fn" rid="fnT1_2">
<sup>b</sup>
</xref>
</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;Non-dominant <xref ref-type="table-fn" rid="fnT1_1">
<sup>a</sup>
</xref>
</td>
<td valign="middle" align="left">72.04 &#xb1; 9.49</td>
<td valign="middle" align="left">72.48 &#xb1; 9.79</td>
<td valign="middle" align="left">0.546 <xref ref-type="table-fn" rid="fnT1_2">
<sup>b</sup>
</xref>
</td>
</tr>
<tr>
<th valign="middle" colspan="4" align="left">Trail making test</th>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;Test A <xref ref-type="table-fn" rid="fnT1_1">
<sup>a</sup>
</xref>
</td>
<td valign="middle" align="left">29.65 &#xb1; 7.78</td>
<td valign="middle" align="left">28.71 &#xb1; 7.50</td>
<td valign="middle" align="left">0.110 <xref ref-type="table-fn" rid="fnT1_2">
<sup>b</sup>
</xref>
</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;Test B <xref ref-type="table-fn" rid="fnT1_1">
<sup>a</sup>
</xref>
</td>
<td valign="middle" align="left">74.69 &#xb1; 27.28</td>
<td valign="middle" align="left">73.42 &#xb1; 25.25</td>
<td valign="middle" align="left">0.527 <xref ref-type="table-fn" rid="fnT1_2">
<sup>b</sup>
</xref>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn id="fnT1_1">
<label>a</label>
<p>Data are presented as mean &#xb1; standard deviation.</p>
</fn>
<fn id="fnT1_2">
<label>b</label>
<p>
<italic>p</italic> by two-tailed independent samples t-test.</p>
</fn>
<fn id="fnT1_3">
<label>c</label>
<p>
<italic>p</italic> by two-tailed Pearson Chi-square test.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Neuropsychological performance was comparable between groups. For the Grooved Pegboard Test, dominant hand completion times were 66.27 &#xb1; 8.45 seconds (alcohol risk) versus 67.27 &#xb1; 9.46 seconds (non-alcohol risk; p=0.153), and non-dominant hand times were 72.04 &#xb1; 9.49 seconds versus 72.48 &#xb1; 9.79 seconds (p=0.546). For the TMT, Part A completion times were 29.65 &#xb1; 7.78 seconds (alcohol risk) versus 28.71 &#xb1; 7.50 seconds (non-alcohol risk; p=0.110), and Part B times were 74.69 &#xb1; 27.28 seconds versus 73.42 &#xb1; 25.25 seconds (p=0.527). Interpreted under these method choices, the absence of significant between-group differences suggests that AUD risk, as defined by screening criteria, may precede measurable neuropsychological deficits in this occupational cohort.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>MRI acquisition</title>
<p>Structural brain MRI scans were acquired using a 3.0 Tesla Philips MRI system (Philips Healthcare, Best, The Netherlands) equipped with a 32-channel head coil. High-resolution three-dimensional T1-weighted images were obtained with the following parameters: repetition time (TR) = 7.4 ms, echo time (TE) = 3.4 ms, flip angle = 8&#xb0;, voxel size = 1 &#xd7; 1 &#xd7; 1 mm&#xb3;, and 180 sagittal slices. All participants were instructed to maintain stillness and neutral head positioning throughout the scanning session to ensure image quality.</p>
</sec>
<sec id="s2_5">
<label>2.5</label>
<title>Data preprocessing</title>
<p>T1-weighted MRI data were preprocessed using the FMRIB Software Library (FSL, version 6.0 (<xref ref-type="bibr" rid="B23">23</xref>) to ensure standardized spatial normalization and artifact minimization. The preprocessing pipeline included both linear and nonlinear registration of each participant&#x2019;s structural MRI to the Montreal Neurological Institute (MNI152) standard space, followed by resampling to a voxel resolution of 2 &#xd7; 2 &#xd7; 2 mm&#xb3;. Following normalization, skull stripping was performed using the High-Definition Brain Extraction Tool (HD-BET), a deep learning&#x2013;based algorithm designed to enhance the accuracy of brain tissue isolation from non-brain elements (<xref ref-type="bibr" rid="B24">24</xref>). This process improved visualization of key anatomical regions, including gray matter, white matter, and ventricular structures, while simultaneously reducing noise and enhancing segmentation fidelity. After skull stripping, the three-dimensional MRI volumes were segmented into 80 two-dimensional axial slices per participant, ensuring standardized anatomical coverage. The use of axial slices provides distinct spatial perspectives and clinically relevant information, facilitating detailed neuroanatomical interpretation and enabling precise detection of structural abnormalities or pathologies (<xref ref-type="bibr" rid="B25">25</xref>, <xref ref-type="bibr" rid="B26">26</xref>).</p>
<p>To increase model generalizability, data augmentation techniques were applied. RandomAffine transformations introduced rotational variations (&#xb1; 10&#xb0;) and translation shifts (&#xb1; 5%). ColorJitter transformations adjusted image brightness and contrast (&#xb1; 20%), and RandomRotation transformations applied further rotational variation (&#xb1; 15&#xb0;) to simulate clinical variability. Pixel intensity normalization was performed using standardized mean and standard deviation values derived from large-scale neuroimaging datasets to standardize input distributions prior to model training (<xref ref-type="bibr" rid="B27">27</xref>&#x2013;<xref ref-type="bibr" rid="B29">29</xref>).</p>
<p>Clinical assessment data underwent systematic preprocessing to ensure data integrity and model compatibility. Missing value analysis was conducted across all clinical variables, including demographic parameters (age, sex), neuropsychological test scores (Grooved Pegboard Test completion times for dominant and non-dominant hands, Trail Making Test Parts A and B), and alcohol use risk indicators (AUDIT scores). Participants with incomplete clinical assessments were excluded from the analytical cohort through listwise deletion, maintaining the methodological rigor requisite for multimodal integration. This conservative approach to missing data management, while potentially reducing statistical power, preserved the validity of cross-modal feature relationships critical to the multimodal learning framework. No imputation strategies were employed to avoid introducing artificial correlations between neuroimaging and clinical features. All continuous clinical variables were retained in their original scales to preserve interpretability, with normalization performed internally within the deep learning architecture through batch normalization layers. Categorical variables, specifically biological sex, were encoded using binary representation (0 = male, 1 = female) consistent with standard practices in medical machine learning applications.</p>
</sec>
<sec id="s2_6">
<label>2.6</label>
<title>Multimodal deep learning framework</title>
<p>To predict alcohol use disorder (AUD) risk in firefighters, we developed a multimodal deep learning framework that integrates structural magnetic resonance imaging (MRI) with clinical and neuropsychological data. The framework comprised three parallel processing branches: a convolutional neural network (CNN) based on ResNet-50 for local morphological feature extraction from MRI images (<xref ref-type="bibr" rid="B30">30</xref>), a Vision Transformer (ViT) module for global contextual representation of neuroanatomical structures (<xref ref-type="bibr" rid="B31">31</xref>), and a multilayer perceptron (MLP) for incorporating clinical and neuropsychological variables (<xref ref-type="bibr" rid="B32">32</xref>). <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> provides a schematic overview of the deep learning architecture, illustrating the parallel processing of MRI images and clinical variables through the ResNet-50, Vision Transformer, and MLP modules, the subsequent feature concatenation, and the final classification layer.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Multimodal deep learning architecture integrating neuroimaging and clinical data for alcohol use disorder risk prediction.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1643552-g002.tif">
<alt-text content-type="machine-generated">Diagram depicting a neural network model for MRI and tabular data processing. The MRI inputs undergo preprocessing, passing through ResNet-50 and Vision Transformer (ViT) components, with features like convolutional blocks and transformer encoders. Tabular data also undergo preprocessing and is processed through a multilayer perceptron (MLP) with fully connected layers and ReLU activations. Features are fused, then enter a fully connected layer, sigmoid activation, and binary classification for final prediction.</alt-text>
</graphic>
</fig>
<p>For the MRI input stream, 80 axial two-dimensional slices per participant were fed into a pretrained ResNet-50 model to derive hierarchical local features. The resulting feature maps underwent average pooling and flattening operations to produce compact representations. In parallel, the same MRI slices were input into the ViT module via patch-based linear embedding. The ViT extracted long-range spatial dependencies and global structural context across brain regions (<xref ref-type="bibr" rid="B33">33</xref>). This dual-path design allowed for the concurrent extraction of both local and global representations from the neuroimaging data.</p>
<p>Simultaneously, clinical and neuropsychological features comprising age, sex, AUDIT score, Grooved Pegboard Test completion times (dominant and non-dominant hand), and Trail Making Test A and B durations were input into an MLP consisting of two fully connected layers with ReLU activations, yielding latent clinical representations. The outputs from the ResNet-50, ViT, and MLP branches were concatenated into a unified feature vector, which was passed through a fully connected layer with a sigmoid activation function to generate a binary prediction of AUD risk. Model training was conducted using the Adam optimizer with an initial learning rate of 0.001, a batch size of 32, and a maximum of 100 training epochs. Early stopping was applied with a patience threshold of 10 epochs based on validation loss. To mitigate overfitting and improve generalizability, dropout regularization (dropout rate = 0.5) was applied to fully connected layers, and batch normalization was incorporated after each convolutional block.</p>
<p>Model evaluation was performed using stratified three-fold cross-validation with participant-level data partitioning to ensure independence between training and validation sets. Performance was assessed based on accuracy, area under the receiver operating characteristic curve (AUROC), sensitivity, and specificity. Statistical differences in AUROC between model configurations were evaluated using DeLong&#x2019;s test (<xref ref-type="bibr" rid="B34">34</xref>). All models were implemented in PyTorch (v1.10) and trained on an NVIDIA RTX A6000 GPU.</p>
</sec>
<sec id="s2_7">
<label>2.7</label>
<title>Model evaluation and statistical analysis</title>
<p>Model performance was evaluated using stratified threefold cross-validation with participant-level data partitioning to ensure independence between training and validation sets. Performance metrics included accuracy, area under the receiver operating characteristic curve (AUROC), precision, and recall. The AUROC was the primary metric due to its robustness to class imbalance. Confidence intervals (95% CI) for AUROC were estimated via bootstrapping (1,000 iterations).</p>
<p>Comparative analyses assessed the multimodal model against unimodal models (MRI-only, clinical-only). Between-model differences in AUROC were tested using DeLong&#x2019;s method, which accounts for the correlation inherent to paired ROC curves evaluated on the same cases (<xref ref-type="bibr" rid="B34">34</xref>). Calibration was evaluated with reliability (calibration) curves to assess agreement between predicted probabilities and observed outcomes, and decision curve analysis was used to quantify net clinical benefit across threshold probabilities relevant to occupational screening. All preprocessing statistics, any calibration fits, and threshold selection were performed within training folds only and applied to the corresponding validation folds to avoid information leakage. Feature importance was analyzed using integrated gradients to enhance interpretability by identifying influential neuroanatomical and clinical inputs contributing to predictions (<xref ref-type="bibr" rid="B35">35</xref>). Statistical analyses were conducted using Python (version 3.9), Scikit-learn (version 1.0), and SciPy (version 1.7).</p>
</sec>
<sec id="s2_8">
<label>2.8</label>
<title>SHapley Additive exPlanations</title>
<p>Feature importance analysis of clinical variables was conducted using SHAP methodology with an XGBoost classifier (<xref ref-type="bibr" rid="B36">36</xref>) trained on clinical features comprising age, sex, AUDIT scores, Grooved Pegboard Test completion times, and Trail Making Test durations. SHAP values were computed using TreeExplainer, which leverages the tree structure for efficient Shapley value calculation (<xref ref-type="bibr" rid="B37">37</xref>). Global feature importance was quantified through mean absolute SHAP values across the cohort, providing interpretable measures of each variable&#x2019;s contribution to risk prediction. Statistical significance of feature contributions was evaluated using permutation-based null hypothesis testing with multiple comparison correction.</p>
</sec>
<sec id="s2_9">
<label>2.9</label>
<title>Gradient-weighted class activation mapping</title>
<p>Gradient-weighted Class Activation Mapping (Grad-CAM) was employed to elucidate the spatial localization of discriminative neuroanatomical features contributing to alcohol use disorder risk classification (<xref ref-type="bibr" rid="B38">38</xref>). This interpretability methodology generates visual explanations by computing the gradient of the predicted class score with respect to the final convolutional layer activations, thereby identifying brain regions that maximally influence the classification decision. The analysis targeted the terminal convolutional layers of each architecture, which preserve spatial resolution while encoding high-level semantic features. The importance weights &#x3b1;<sub>k</sub>
<sup>c</sup> for each feature map k with respect to target class c were computed through gradient backpropagation:</p>
<disp-formula>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>c</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>Z</mml:mi>
</mml:mfrac>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>i</mml:mi>
</mml:munder>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>j</mml:mi>
</mml:munder>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mi>c</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where y<sup>c</sup> denotes the class score, A<sup>k</sup>
<sub>ij</sub> represents the activation at spatial location (i,j) in feature map k, and Z normalizes by spatial dimensions. The final class-discriminative localization map was generated through weighted combination of forward activation maps:</p>
<disp-formula>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msubsup>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>A</mml:mi>
<mml:mi>M</mml:mi>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>k</mml:mi>
</mml:munder>
<mml:msubsup>
<mml:mi>a</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>c</mml:mi>
</mml:msubsup>
<mml:msup>
<mml:mi>A</mml:mi>
<mml:mi>k</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The resulting coarse-grained heatmaps underwent bilinear interpolation to match the original image resolution (224&#xd7;224 pixels) and were superimposed on the corresponding MRI slices with a transparency coefficient of 0.6 to facilitate anatomical interpretation. Visualizations were generated for a randomly selected subset of 50 participants per risk category to assess spatial consistency of learned features. Dice similarity coefficients quantified the spatial overlap of activation patterns across participants, while occlusion sensitivity analysis validated the causal importance of identified regions by measuring classification confidence degradation upon masking the upper quintile of activation intensities. Given these implementation details, we briefly justify our choice of localization method. We selected Grad-CAM after considering alternative feature-localization techniques because it is class-discriminative, CNN-architecture agnostic, and computationally efficient for 2D multi-slice MRI. Unlike vanilla saliency, which is high-variance and visually noisy, Grad-CAM yields stable, coarse-to-mid-scale heatmaps aligned with the target class. Integrated Gradients requires a baseline and path integral whose choice is non-trivial for T1 intensity scales and can introduce baseline-dependent artifacts, whereas Grad-CAM avoids a baseline choice while remaining faithful to score&#x2013;gradient information. Occlusion/perturbation and LIME/SHAP image explanations impose heavy sampling costs and design choices (e.g., patch size, superpixels) that scale poorly to ~80 slices per subject (<xref ref-type="bibr" rid="B37">37</xref>, <xref ref-type="bibr" rid="B39">39</xref>). Transformer attention maps are not inherently class-specific and may not reflect decision-critical evidence, whereas Grad-CAM is explicitly class-discriminative. In medical imaging, Grad-CAM&#x2019;s regional localization aligns with radiological reading practices, enabling transparent overlays on axial slices and cohort-level aggregation without re-training. To address known limitations (resolution tied to the last conv layer), we performed sanity checks (parameter randomization and slice-wise ablation) and report both representative and aggregated maps (<xref ref-type="bibr" rid="B40">40</xref>).</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results</title>
<p>
<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> summarizes the comparative performance of multiple predictive models for alcohol use disorder (AUD) risk classification, including clinical-only, neuroimaging-only, multi-scale image integration, and multimodal models integrating neuroimaging and clinical data. Metrics include accuracy, area under the receiver operating characteristic curve (AUROC), precision, and recall were evaluated using stratified threefold cross-validation to ensure robust estimates.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Performance comparison of alcohol use disorder risk prediction models.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Model</th>
<th valign="middle" align="left">Accuracy</th>
<th valign="middle" align="left">AUROC</th>
<th valign="middle" align="left">Precision</th>
<th valign="middle" align="left">Recall</th>
</tr>
</thead>
<tbody>
<tr>
<th valign="middle" colspan="5" align="left">Clinical only</th>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;Logistic Regression</td>
<td valign="middle" align="left">0.6253</td>
<td valign="middle" align="left">0.5773</td>
<td valign="middle" align="left">0.6117</td>
<td valign="middle" align="left">0.6432</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;MLP</td>
<td valign="middle" align="left">0.5637</td>
<td valign="middle" align="left">0.5436</td>
<td valign="middle" align="left">0.5521</td>
<td valign="middle" align="left">0.5748</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;Random Forest</td>
<td valign="middle" align="left">0.5487</td>
<td valign="middle" align="left">0.5388</td>
<td valign="middle" align="left">0.5294</td>
<td valign="middle" align="left">0.5562</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;XGBoost</td>
<td valign="middle" align="left">0.4857</td>
<td valign="middle" align="left">0.4795</td>
<td valign="middle" align="left">0.4783</td>
<td valign="middle" align="left">0.4620</td>
</tr>
<tr>
<th valign="middle" colspan="5" align="left">Image only</th>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;ResNet50</td>
<td valign="middle" align="left">0.6153</td>
<td valign="middle" align="left">0.5773</td>
<td valign="middle" align="left">0.6089</td>
<td valign="middle" align="left">0.6241</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;EfficientNet-B0</td>
<td valign="middle" align="left">0.5967</td>
<td valign="middle" align="left">0.5648</td>
<td valign="middle" align="left">0.5891</td>
<td valign="middle" align="left">0.6034</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;ViT</td>
<td valign="middle" align="left">0.5457</td>
<td valign="middle" align="left">0.5395</td>
<td valign="middle" align="left">0.5412</td>
<td valign="middle" align="left">0.5480</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;DeiT</td>
<td valign="middle" align="left">0.5037</td>
<td valign="middle" align="left">0.5236</td>
<td valign="middle" align="left">0.5001</td>
<td valign="middle" align="left">0.5076</td>
</tr>
<tr>
<th valign="middle" colspan="5" align="left">Multi-scale image</th>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;ResNet50 + ViT</td>
<td valign="middle" align="left">0.6354</td>
<td valign="middle" align="left">0.6187</td>
<td valign="middle" align="left">0.6213</td>
<td valign="middle" align="left">0.6495</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;ResNet50 + DeiT</td>
<td valign="middle" align="left">0.5833</td>
<td valign="middle" align="left">0.5325</td>
<td valign="middle" align="left">0.5702</td>
<td valign="middle" align="left">0.5894</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;EfficientNet-B0 + ViT</td>
<td valign="middle" align="left">0.5902</td>
<td valign="middle" align="left">0.5869</td>
<td valign="middle" align="left">0.5820</td>
<td valign="middle" align="left">0.5981</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;EfficientNet-B0+ DeiT</td>
<td valign="middle" align="left">0.5627</td>
<td valign="middle" align="left">0.5398</td>
<td valign="middle" align="left">0.5514</td>
<td valign="middle" align="left">0.5739</td>
</tr>
<tr>
<th valign="middle" colspan="5" align="left">Multimodal (image + clinical)</th>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;ResNet50 + ViT + MLP</td>
<td valign="middle" align="left">
<bold>0.7988</bold>
</td>
<td valign="middle" align="left">
<bold>0.7965</bold>
</td>
<td valign="middle" align="left">
<bold>0.7836</bold>
</td>
<td valign="middle" align="left">
<bold>0.8124</bold>
</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;ResNet50 + ViT + LR</td>
<td valign="middle" align="left">0.6887</td>
<td valign="middle" align="left">0.6726</td>
<td valign="middle" align="left">0.6752</td>
<td valign="middle" align="left">0.6989</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;ResNet50 + DeiT + MLP</td>
<td valign="middle" align="left">0.7726</td>
<td valign="middle" align="left">0.7563</td>
<td valign="middle" align="left">0.7590</td>
<td valign="middle" align="left">0.7852</td>
</tr>
<tr>
<td valign="middle" align="left">&#x2003;EfficientNet-B0+ DeiT + LR</td>
<td valign="middle" align="left">0.6429</td>
<td valign="middle" align="left">0.6854</td>
<td valign="middle" align="left">0.6381</td>
<td valign="middle" align="left">0.6587</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Performance metrics (Accuracy, AUROC, Precision, Recall) for alcohol use disorder risk prediction models across four architectural categories: clinical data only models, neuroimaging only architectures, multi scale image integration approaches, and multimodal frameworks combining neuroimaging with clinical variables. Bold values indicate the highest performance metrics across all evaluated models.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Among clinical-only models, logistic regression yielded the best performance, achieving an accuracy of 62.53%, AUROC of 57.73%, precision of 61.17%, and recall of 64.32%. Other clinical models, including multilayer perceptron (MLP), random forest, and XGBoost, demonstrated lower classification accuracy and area under the curve (AUC), with AUROCs ranging from 47.95% to 54.36%.</p>
<p>In the neuroimaging-only condition, the ResNet-50 model outperformed other architectures such as EfficientNet-B0, Vision Transformer (ViT), and Data-efficient Image Transformer (DeiT), achieving an AUROC of 57.73% and an accuracy of 61.53%. The ViT and DeiT models yielded AUROCs of 53.95% and 52.36%, respectively, suggesting that these transformer-based models did not surpass the convolutional baseline in unimodal imaging tasks.</p>
<p>Combining multiple image architectures slightly improved performance. The ResNet-50 + ViT hybrid configuration achieved the highest AUROC (61.87%) and accuracy (63.54%) within the multi-scale image category. Nonetheless, performance remained suboptimal compared to multimodal approaches.</p>
<p>The multimodal frameworks that integrated both neuroimaging and clinical data demonstrated significant improvements in predictive performance. The optimal configuration consisted of a fusion architecture incorporating ResNet-50, ViT, and an MLP for clinical variables, which achieved an accuracy of 79.88%, AUROC of 79.65%, precision of 78.36%, and recall of 81.24%. This performance represents a 17.35 percentage point improvement in accuracy and a 21.92 percentage gain in AUROC over the best clinical-only model (logistic regression), thereby providing compelling evidence for the additive benefit of multimodal integration. Other multimodal variants such as ResNet-50 + DeiT + MLP and ResNet-50 + ViT + logistic regression also showed superior performance relative to unimodal baselines but did not match the top-performing model.</p>
<p>Statistical comparison of AUROC values using DeLong&#x2019;s test confirmed that the multimodal ResNet-50 + ViT + MLP model significantly outperformed both clinical-only and image-only models (p&lt; 0.001).</p>
<p>
<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> provides a visual representation of model performance across three complementary dimensions. The ROC curves (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3A</bold>
</xref>) demonstrate a clear separation between the multimodal architecture and other modeling approaches, with the multimodal curve exhibiting a substantially greater area under the curve (AUC). The multi-scale image model shows intermediate discriminative capacity, positioned between the multimodal framework and the unimodal approaches, which demonstrate comparable but less robust discriminative performance.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Comparison of model performance for alcohol use disorder risk prediction across data modalities. <bold>(A)</bold> Receiver operating characteristic (ROC) curves illustrate the discriminative performance of clinical-only, image-only, multi-scale image, and multimodal models, with the multimodal model showing the highest area under the curve (AUC). <bold>(B)</bold> Calibration curves compare predicted versus observed probabilities, demonstrating superior calibration in the multimodal model. <bold>(C)</bold> Decision curve analysis indicates that the multimodal model provides the greatest net clinical benefit across a range of threshold probabilities.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1643552-g003.tif">
<alt-text content-type="machine-generated">Three sets of graphs. (A) ROC curves showing true positive rate versus false positive rate for clinical only, image only, multi-scale image, and multimodal data. (B) Calibration curves illustrating fraction of positive versus mean predicted probability in the same categories. (C) Decision curves depicting net benefit versus threshold probability for each category. Each set uses different colors and includes multiple machine learning models for comparison.</alt-text>
</graphic>
</fig>
<p>Calibration curves (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3B</bold>
</xref>) reveal that the multimodal approach aligns more closely with the ideal calibration line compared to alternative models. Clinical-only and neuroimaging-only approaches exhibit noticeable deviations from optimal calibration, particularly in lower probability regions where systematic overestimation is visually apparent.</p>
<p>Decision curve analysis (<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3C</bold>
</xref>) illustrates that the multimodal framework provides a consistently positive net benefit across a broader range of threshold probabilities relative to other modeling strategies. In contrast, clinical-only and neuroimaging-only approaches show diminished clinical utility at higher threshold values, whereas the multimodal approach maintains its net benefit across the full probability spectrum. These visual assessments corroborate the quantitative findings presented in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>, further supporting the enhanced predictive capability achieved through multimodal integration.</p>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> shows confusion matrices for representative models (clinical-only Logistic Regression; image-only ResNet-50; multi-scale image ResNet-50 + ViT; multimodal ResNet-50 + ViT + MLP). The multimodal model yielded TN = 55, FP = 15, FN = 13, TP = 54. The clinical-only model produced TN = 44, FP = 26, FN = 25, TP = 42. The image-only model produced TN = 43, FP = 27, FN = 25, TP = 42. The multi-scale image model produced TN = 43, FP = 27, FN = 23, TP = 44. Overall, the multimodal configuration simultaneously reduced both FP and FN relative to the other approaches, indicating a more favorable error profile for occupational screening.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Confusion matrices comparing classification performance across model architectures for alcohol use disorder risk prediction. <bold>(A)</bold> Clinical-only model using logistic regression. <bold>(B)</bold> Image-only model using ResNet-50. <bold>(C)</bold> Multi-scale image model combining ResNet-50 and Vision Transformer. <bold>(D)</bold> Multimodal model integrating ResNet-50, Vision Transformer, and clinical variables through MLP. Values represent the number of participants classified in each category from stratified 3-fold cross-validation.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpsyt-16-1643552-g004.tif">
<alt-text content-type="machine-generated">Four confusion matrices labeled A, B, C, and D, compare different models. (A) Logistic Regression: 44 normal, 26 misclassified; 42 alcohol risk, 25 misclassified. (B) ResNet50: 43 normal, 27 misclassified; 42 alcohol risk, 25 misclassified. (C) ResNet50 + ViT: 43 normal, 27 misclassified; 44 alcohol risk, 23 misclassified. (D) ResNet50 + ViT + MLP: 55 normal, 15 misclassified; 54 alcohol risk, 13 misclassified. Color gradients show classification accuracy.</alt-text>
</graphic>
</fig>
<p>To examine the feature extraction patterns of neuroimaging-only models, we performed gradient-weighted class activation mapping (Grad-CAM) analysis on both ResNet-50 and EfficientNet-B0 architectures. <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref> displays representative Grad-CAM visualizations of axial brain slices from randomly selected participants processed through these convolutional neural network models. The Grad-CAM activations from both architectures exhibited substantial spatial heterogeneity across slices. Activation intensities showed irregular distributions, with discrete focal hotspots in some regions and diffuse low-intensity patterns across broader anatomical areas. Both ResNet-50 and EfficientNet-B0 revealed no systematic concentration within specific neuroanatomical structures, with high-intensity regions appearing stochastically distributed across cortical and subcortical territories. Peak activation values varied markedly across slices and architectures, ranging from isolated punctate foci to broad activation zones encompassing multiple anatomical regions.</p>
<p>To further justify the use of Grad-CAM over alternative feature localization methods, we additionally applied Vanilla Saliency (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref>), Integrated Gradients (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3</bold>
</xref>), and Occlusion Sensitivity (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S4</bold>
</xref>). Compared with Grad-CAM, Vanilla Saliency and Integrated Gradients produced noisy, low-contrast attribution maps with limited anatomical interpretability, consistent with known limitations of these gradient-based approaches when applied to structural MRI data. Occlusion Sensitivity yielded block-like activation patterns resulting from the perturbation grid but failed to delineate neuroanatomically meaningful regions in a stable manner. In contrast, Grad-CAM consistently generated smoother and more interpretable overlays, aligning with prior reports demonstrating its robustness and clinical plausibility in neuroimaging. These supplementary comparisons underscore the suitability of Grad-CAM as the primary visualization approach in this study.</p>
<p>These visualization outputs corroborate the quantitative performance metrics observed for the neuroimaging-only models (ResNet-50 AUROC: 57.73%, accuracy: 61.53%; EfficientNet-B0 AUROC: 56.54%, accuracy: 60.82%). The absence of consistent activation patterns across the randomly sampled cases in both architectures provides empirical evidence for the limited feature extraction capability of image-only models in this AUD risk prediction task.</p>
<p>Feature importance analysis of the clinical variables was conducted using SHapley Additive exPlanations (SHAP) to quantify individual feature contributions to the multimodal model predictions. <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref> presents the SHAP value distributions for six clinical features incorporated in the optimal multimodal framework.</p>
<p>The SHAP analysis revealed differential feature contributions across the clinical variable set. Sex demonstrated the most pronounced positive impact on model predictions, with SHAP values ranging from approximately -0.05 to +0.45, exhibiting a strong rightward skew. Non-dominant hand Grooved Pegboard completion times (GP_nondom_sec_adj) displayed bidirectional effects with SHAP values distributed between -0.30 and +0.25, indicating variable contributions to risk prediction depending on individual performance levels.</p>
<p>Age exhibited a balanced distribution of SHAP values spanning -0.25 to +0.20, with the majority of instances clustering near zero. Dominant hand Grooved Pegboard performance (GP_dom_sec_adj) showed similar bidirectional patterns with values ranging from -0.20 to +0.15. Trail Making Test Part A completion times (TrailA_time_adj) demonstrated moderate feature importance with SHAP values between -0.15 and +0.15. Trail Making Test Part B completion times (TrailB_time_adj) yielded the most concentrated distribution around zero, with limited outliers extending to &#xb1;0.20, suggesting minimal direct contribution to prediction outcomes in the multimodal context.</p>
<p>These quantitative feature attribution results complement the multimodal model performance metrics, providing mechanistic insights into the relative contributions of individual clinical variables within the integrated predictive framework.</p>
</sec>
<sec id="s4" sec-type="discussion">
<label>4</label>
<title>Discussion</title>
<p>The multimodal deep learning framework demonstrated superior classification performance, validating the synergistic integration of structural neuroimaging with clinical assessments for AUD risk stratification in firefighters. The principal findings encompass: (1) The synergistic combination of ResNet-50 and Vision Transformer architectures facilitates complementary extraction of local morphological features and global spatial dependencies from structural MRI data, obviating computationally intensive functional connectivity analyses; (2) Integration of standardized neuropsychological assessments, specifically the Grooved Pegboard Test and Trail Making Test, provides functional neurological proxies that partially compensate for the absence of task-based or resting-state functional MRI data; (3) The multimodal framework demonstrates a 17.35 percentage point improvement in classification accuracy relative to clinical-only models, substantiating the discriminative value of structural neuroimaging when appropriately integrated with behavioral metrics; (4) Feature importance analysis identifies sex as the predominant clinical predictor, followed by motor coordination measures, elucidating potential sex-specific vulnerability patterns within this occupational cohort; (5) The model maintains robust calibration across probability thresholds, suggesting clinical applicability for risk stratification without the operational complexity inherent to functional neuroimaging protocols.</p>
<sec id="s4_1">
<label>4.1</label>
<title>Comparative analysis with extant literature</title>
<p>
<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> provides a comprehensive summary of recent multimodal deep learning approaches for psychiatric disorder prediction, contextualizing our findings within the broader landscape of neuroimaging-based classification studies. Contemporary neuroimaging investigations have consistently demonstrated the superiority of multimodal approaches combining structural MRI, functional task-based MRI, and resting-state functional connectivity in psychiatric classification tasks (<xref ref-type="bibr" rid="B46">46</xref>). However, the multimodal framework presented herein, achieving 79.88% accuracy (AUROC: 0.795) through structural MRI and clinical assessment integration alone, demonstrates competitive performance relative to architectures incorporating functional neuroimaging. A recent triple-modality integration study (<xref ref-type="bibr" rid="B42">42</xref>) (sMRI + fMRI + SNP) yielded 79.01% accuracy in schizophrenia classification (n=492), with individual modalities contributing differentially (sMRI: 66.33%, fMRI: 75.29%, SNP: 57.06%). The marginal improvement from sMRI baseline to multimodal integration (13.68 percentage points) must be contextualized against substantially increased acquisition complexity and computational burden. A comprehensive investigation utilizing an extensive neuroimaging battery comprising 119 alcohol-dependent patients and 97 controls revealed that while multimodal integration yielded optimal classification performance, the investigators concluded that &#x201c;in terms of direct clinical applicability, currently the most realistic neuroimaging-based classifier for AD may be unimodal based on structural MRI and grey-matter density specifically&#x201d; (<xref ref-type="bibr" rid="B46">46</xref>), citing the temporal demands and analytical complexity of functional MRI protocols. This empirical observation corroborates our methodological decision to prioritize T1-weighted structural MRI as the primary neuroimaging modality.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparative summary of multimodal deep learning approaches for psychiatric disorder prediction.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="left">Study</th>
<th valign="middle" align="left">Modalities</th>
<th valign="middle" align="left">Target disease</th>
<th valign="middle" align="left">Accuracy</th>
<th valign="middle" align="left">AUROC</th>
<th valign="middle" align="left">Sample size</th>
<th valign="middle" align="left">Sample characteristic</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Zheng et&#xa0;al. (<xref ref-type="bibr" rid="B41">41</xref>)</td>
<td valign="middle" align="left">sMRI + fMRI</td>
<td valign="middle" align="left">MDD</td>
<td valign="middle" align="left">75.2%</td>
<td valign="middle" align="left">0.808</td>
<td valign="middle" align="left">2319</td>
<td valign="middle" align="left">Control<break/>MDD</td>
</tr>
<tr>
<td valign="middle" align="left">Kanyal et&#xa0;al. (<xref ref-type="bibr" rid="B42">42</xref>)</td>
<td valign="middle" align="left">sMRI + fMRI + SNP</td>
<td valign="middle" align="left">SZ</td>
<td valign="middle" align="left">79.01%</td>
<td valign="middle" align="left">&#x2013;</td>
<td valign="middle" align="left">492</td>
<td valign="middle" align="left">Control<break/>SZ</td>
</tr>
<tr>
<td valign="middle" align="left">Zhu et&#xa0;al. (<xref ref-type="bibr" rid="B43">43</xref>)</td>
<td valign="middle" align="left">sMRI (3D) + fMRI</td>
<td valign="middle" align="left">AUD</td>
<td valign="middle" align="left">67.4% -90.5%</td>
<td valign="middle" align="left">&#x2013;</td>
<td valign="middle" align="left">92</td>
<td valign="middle" align="left">Control<break/>AUD</td>
</tr>
<tr>
<td valign="middle" align="left">Vergara et&#xa0;al. (<xref ref-type="bibr" rid="B44">44</xref>)</td>
<td valign="middle" align="left">fMRI</td>
<td valign="middle" align="left">AUD</td>
<td valign="middle" align="left">&#x2013;</td>
<td valign="middle" align="left">0.79</td>
<td valign="middle" align="left">102</td>
<td valign="middle" align="left">Control<break/>AUD</td>
</tr>
<tr>
<td valign="middle" align="left">Kamarajan et&#xa0;al. (<xref ref-type="bibr" rid="B45">45</xref>)</td>
<td valign="middle" align="left">fMRI</td>
<td valign="middle" align="left">AUD</td>
<td valign="middle" align="left">76.67%</td>
<td valign="middle" align="left">0.93</td>
<td valign="middle" align="left">60</td>
<td valign="middle" align="left">Control<break/>AUD<break/>Male participants only</td>
</tr>
<tr>
<td valign="middle" align="left">Guggenmos et&#xa0;al. (<xref ref-type="bibr" rid="B46">46</xref>)</td>
<td valign="middle" align="left">sMRI + fMRI</td>
<td valign="middle" align="left">AUD</td>
<td valign="middle" align="left">79.3%</td>
<td valign="middle" align="left">&#x2013;</td>
<td valign="middle" align="left">216</td>
<td valign="middle" align="left">Control<break/>AUD</td>
</tr>
<tr>
<td valign="middle" align="left">Ours</td>
<td valign="middle" align="left">
<bold>sMRI (2D)</bold> + <bold>Clinical</bold>
</td>
<td valign="middle" align="left">
<bold>AUD</bold>
</td>
<td valign="middle" align="left">
<bold>79.88%</bold>
</td>
<td valign="middle" align="left">
<bold>0.795</bold>
</td>
<td valign="middle" align="left">
<bold>689</bold>
</td>
<td valign="middle" align="left">
<bold>Occupational (Firefighters)</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Reported values are taken from the cited papers; numbers are not directly comparable across studies because of differences in datasets, label definitions (diagnosis vs. risk), cohort composition, scanners/protocols, and evaluation procedures (cross-validation vs. held-out tests). When a study reported multiple results, we list a representative value or a range; &#x201c;-&#x201d; indicates the metric was not reported.</p>
</fn>
<fn>
<p>Modalities: sMRI, T1-weighted structural MRI; fMRI, functional MRI (resting or task-based as reported); SNP, single-nucleotide polymorphisms.</p>
</fn>
<fn>
<p>Target disease: MDD, major depressive disorder; SZ, schizophrenia; AUD, alcohol use disorder.</p>
</fn>
<fn>
<p>Sample size is the total N analyzed in each study; Sample characteristic summarizes comparison groups (e.g., control vs. disorder, sex restrictions).</p>
</fn>
<fn>
<p>Ours denotes an occupational firefighter cohort and a multimodal model using sMRI (2D axial slices) + clinical variables without fMRI; results are averaged over stratified 3-fold, subject-wise cross-validation.</p>
</fn>
<fn>
<p>Bold values represent results from the current study.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Previous investigations employing resting-state functional connectivity have reported classification accuracies ranging from 61.53% to 76.67% for discriminating alcohol-dependent individuals from controls (<xref ref-type="bibr" rid="B43">43</xref>, <xref ref-type="bibr" rid="B44">44</xref>). Direct comparison with AUD-focused investigations reveals our framework&#x2019;s competitive performance despite methodological parsimony. A recent study (<xref ref-type="bibr" rid="B44">44</xref>)reported resting-state fMRI yielding AUROC 0.79 (n=102), comparable to our 0.795 despite utilizing computationally intensive connectivity analyses. This equivalence challenges assumptions regarding the superior discriminative capacity of functional imaging for AUD detection. Similarly, another investigation (<xref ref-type="bibr" rid="B46">46</xref>) demonstrated that dual neuroimaging modality integration (sMRI + fMRI) achieved 79.3% accuracy in AUD classification (n=216), representing merely 2.7 percentage points improvement over single modality (76.6%) which represents a limited enhancement that raises critical questions regarding the cost-effectiveness of functional imaging protocols in occupational screening contexts. Random Forest classification leveraging functional connectivity within the Default Mode Network combined with neuropsychological measures achieved 76.67% accuracy (<xref ref-type="bibr" rid="B45">45</xref>), necessitating extensive preprocessing pipelines and network-level analytical frameworks. Recent work (<xref ref-type="bibr" rid="B45">45</xref>) reported fMRI-based classification achieving 76.67% accuracy (AUROC: 0.93) in a male-exclusive cohort (n=60). Our superior accuracy (79.88%) in a substantially larger sample (n=689) with mixed-gender composition suggests that structural alterations combined with behavioral assessments may provide greater discriminative capacity than functional connectivity alone in occupational populations. Resting-state connectivity features have demonstrated capacity to explain 33% of variance in Alcohol Use Disorders Identification Test (AUDIT) scores (<xref ref-type="bibr" rid="B47">47</xref>), though such models required acquisition of multiple functional MRI sequences including monetary incentive delay and face-matching paradigms alongside resting-state protocols.</p>
<p>Multimodal data integration approaches in psychiatric research demonstrate methodological advantages comparable to our neuroimaging-clinical framework. A recent VR-based study (<xref ref-type="bibr" rid="B48">48</xref>) developed machine learning models utilizing acoustic and physiological features VR exposure sessions for social anxiety disorder, achieving an AUROC of 0.852 with CatBoost for Social Phobia Scale prediction using multimodal features (n=132 samples from 25 participants). Notably, their analysis revealed that acoustic features (AUROC: 0.788) substantially outperformed physiological features alone (AUROC: 0.626) for anxiety symptom prediction, with multimodal integration yielding superior classification performance across multiple anxiety domains. While their VR-based approach differs methodologically from our structural neuroimaging framework, the 7.26 percentage point improvement from physiological to multimodal features (compared to our 17.35 percentage point improvement from clinical to multimodal) highlights the consistent benefit of cross-modal integration in psychiatric risk stratification. Their findings that acoustic biomarkers captured more discriminative information than physiological responses during anxiety-inducing scenarios parallels our observation that targeted neuropsychological assessments provide critical functional anchoring for structural alterations.</p>
<p>The classification performance of our T1-weighted structural MRI multimodal approach (79.88% accuracy) demonstrates favorable comparison with functional connectivity-based methodologies while offering considerable practical advantages regarding acquisition efficiency and computational parsimony. Our framework&#x2019;s 17.35 percentage point improvement from clinical-only baseline (62.53%) substantially exceeds the incremental gains observed when adding neuroimaging to clinical data reported previously (<xref ref-type="bibr" rid="B46">46</xref>), suggesting that targeted neuropsychological assessments may capture variance typically attributed to functional connectivity measures. A systematic review examining machine learning applications in AUD reported neuroimaging-based algorithms achieving sensitivity ranging from 90-99.99% and specificity from 82-99.97% (<xref ref-type="bibr" rid="B14">14</xref>); however, these exceptional performance metrics were predominantly observed in investigations combining multiple imaging modalities. One study (<xref ref-type="bibr" rid="B43">43</xref>)reported 3D sMRI + fMRI combination achieving accuracy ranging from 67.4% to 90.5% (n=92). The substantial variability suggests potential overfitting in small samples, emphasizing the importance of our larger cohort (n=689) for robust generalization estimates. Beyond structural neuroimaging approaches, recent advances in machine learning applications for mental health monitoring in first responders provide complementary perspectives on psychological distress prediction. A proof-of-concept investigation (<xref ref-type="bibr" rid="B49">49</xref>) developed predictive models for posttraumatic stress injuries (PTSI) utilizing intensive longitudinal data from 274 Montreal firefighters monitored biweekly across 12 weeks. The study implemented four distinct machine learning algorithms (logistic regression, support vector classifier, extreme gradient boosting) trained on temporal sequences of standardized psychological assessments (PHQ-9, GAD-7, PCL-5) and psychosocial variables (occupational stress, social support, coping strategies). The optimal model configuration, employing extreme gradient boosting with three lagged measurement timepoints and comprehensive feature sets, achieved 94% classification accuracy (AUC = 0.93, sensitivity = 0.61, specificity = 0.97). Several methodological contrasts with the present investigation merit consideration. The documented PTSI prevalence, fluctuating between 6.9% and 10.6% across assessment intervals with cumulative incidence of 19.7%, represents substantially lower psychopathology rates than our observed AUD risk prevalence of 56.9%, potentially attributable to differential diagnostic thresholds between acute stress-related symptomatology and chronic alcohol use vulnerability. Feature importance analyses identified lagged PHQ-9 scores collected 2 and 6 weeks prior to target assessment as dominant predictors (19% and 10% relative importance respectively), with GAD-7 and PCL-5 scores contributing secondarily, while demographic variables (age &gt;46 years, work experience &gt;21 years) demonstrated minimal predictive value. This hierarchical pattern corresponds with our SHAP-derived feature attributions wherein neuropsychological performance metrics superseded demographic characteristics. The temporal dependency of predictive accuracy, wherein model performance systematically improved from single-timepoint (accuracy range: 0.81-0.91) to three-timepoint configurations (accuracy range: 0.82-0.94), underscores the critical importance of longitudinal symptom trajectories in psychiatric risk modeling. These convergent findings across distinct methodological paradigms substantiate the superiority of multimodal, temporally-informed approaches over cross-sectional univariate assessments. Integration of periodic structural neuroimaging for baseline vulnerability characterization with continuous smartphone-based symptom monitoring could potentially optimize early intervention strategies through synthesis of stable neurobiological markers and dynamic clinical trajectories. Our approach achieves minimal sacrifice in predictive accuracy while greatly reducing acquisition time, computational burden, and technical expertise requirements (<xref ref-type="bibr" rid="B49">49</xref>).</p>
<p>Previous investigations utilizing isolated structural MRI modalities have provided valuable performance benchmarks, with grey matter density analysis achieving 65% classification accuracy in comprehensive multimodal comparisons (<xref ref-type="bibr" rid="B46">46</xref>). The present study builds upon these findings by demonstrating that augmenting structural neuroimaging with targeted neuropsychological assessments yields enhanced discriminative capacity (79.88% accuracy), consistent with theoretical frameworks positing synergistic information capture across neurobiological and behavioral domains. The 17.35 percentage point improvement from clinical baseline reflects fundamental complementarity rather than simple feature concatenation: structural neuroimaging captures cumulative morphological alterations reflecting chronic alcohol exposure, providing stable biomarkers of neurotoxic burden, while neuropsychological performance offers dynamic functional readouts of neural system integrity sensitive to subclinical impairments. This performance differential underscores the critical importance of incorporating standardized neuropsychological assessments to compensate for the absence of functional connectivity information. Recent investigations have emphasized that machine learning algorithms provide valuable tools for quantifying large-scale network differences in AUD (<xref ref-type="bibr" rid="B44">44</xref>); however, our results suggest that morphological features combined with targeted clinical assessments achieve comparable discriminative capacity.</p>
<p>The firefighter population presents unique challenges for AUD prediction modeling. While general population studies have examined heterogeneous samples characterized by varied substance use histories and psychiatric comorbidities (<xref ref-type="bibr" rid="B43">43</xref>), our cohort&#x2019;s occupational homogeneity and elevated baseline risk necessitated tailored analytical approaches. With n=689, our investigation represents the second-largest cohort among reviewed studies [following a recent MDD study (<xref ref-type="bibr" rid="B41">41</xref>): n=2319], providing robust statistical power while maintaining occupational homogeneity. While T1-weighted structural sequences and fMRI share similar acquisition times (5&#x2013;10 minutes each), the critical distinction lies in post-processing complexity. Functional MRI necessitates sophisticated preprocessing pipelines encompassing motion correction, temporal filtering, spatial smoothing, and connectivity analysis, extending analysis time from hours to days. Additionally, fMRI&#x2019;s heightened motion sensitivity increases data attrition rates, compromising practicality for large-scale screening initiatives. Previous occupational cohort investigations remain limited, constraining direct performance comparisons. Nevertheless, the effective classification achieved without functional MRI suggests that structural alterations and behavioral manifestations may exhibit enhanced discriminability in high-risk occupational groups, potentially attributable to chronic stress exposure and cultural factors influencing alcohol consumption patterns. The elimination of fMRI-specific infrastructure requirements (stimulus presentation systems, synchronization hardware, specialized preprocessing software) substantially reduces implementation barriers in clinical settings, supporting the translational feasibility of our approach for occupational health surveillance.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Mechanistic considerations</title>
<p>The efficacy of our multimodal approach necessitates examination through complementary interpretability methodologies to elucidate differential contributions of neuroimaging and clinical features. Recent advances in multimodal explainable AI have demonstrated the critical importance of understanding feature interactions across modalities. A recent study achieved 94.81% accuracy using an Ensemble Optimization-enabled Explainable CNN (EO-ECNN) with multimodal data integration, highlighting the significance of interpretability in clinical applications (<xref ref-type="bibr" rid="B50">50</xref>). Recent comparative studies have systematically evaluated various explainability approaches for multimodal medical imaging. A large-scale experiment across four medical imaging datasets found that while attention maps from Vision Transformers generally surpass Grad-CAM in explainability, transformer-specific interpretability methods demonstrate superior performance (<xref ref-type="bibr" rid="B51">51</xref>). This finding underscores the importance of selecting architecture-appropriate interpretability techniques rather than applying traditional CNN-based methods to transformer architecture. Gradient-weighted Class Activation Mapping (Grad-CAM) analysis applied to unimodal neuroimaging models revealed critical insights regarding the limitations of structural MRI-only approaches. As illustrated in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>, Grad-CAM visualizations from both ResNet-50 (Panel A) and EfficientNet-B0 (Panel B) architectures demonstrated stochastic activation patterns across randomly selected axial brain slices. The activation maps exhibited no systematic concentration within anatomically relevant regions associated with alcohol-related neurodegeneration, instead displaying diffuse, heterogeneous patterns with focal hotspots appearing randomly across cortical and subcortical territories. The stochastic activation patterns observed through Grad-CAM analysis provide empirical evidence for the fundamental limitations of structural MRI-only approaches in detecting subtle, distributed alterations associated with AUD risk. This finding aligns with previous neuroimaging studies showing that morphological changes in early-stage AUD are often diffuse and heterogeneous, requiring behavioral anchoring for meaningful interpretation (<xref ref-type="bibr" rid="B46">46</xref>, <xref ref-type="bibr" rid="B52">52</xref>).</p>
<p>This absence of neuroanatomically coherent feature extraction in image-only models provides mechanistic validation for observed performance limitations (ResNet-50: 57.73% AUROC; EfficientNet-B0: 56.54% AUROC). A comprehensive survey of explainable multimodal learning methods confirmed that such random activation patterns indicate insufficient discriminative capacity when structural alterations are subtle and distributed (<xref ref-type="bibr" rid="B53">53</xref>). Recent advances in transformer architecture have introduced attention visualization as a complementary interpretability approach. Studies on multimodal foundation models for anomaly detection have demonstrated that combining SHAP, Grad-CAM, and attention visualization provides more comprehensive insights than any single approach, particularly when dealing with heterogeneous medical data sources (<xref ref-type="bibr" rid="B54">54</xref>). These findings suggest that different XAI techniques capture complementary aspects of model behavior: spatial localization through Grad-CAM, feature importance through SHAP, and hierarchical relationships through attention mechanisms. Grad-CAM heatmaps revealed that convolutional neural networks, when constrained to structural MRI data alone, failed to converge on consistent morphological markers despite well-established volumetric alterations in alcohol-dependent populations. Peak activation intensities varied markedly between slices without correspondence to regions of established vulnerability including prefrontal cortex, hippocampus, or cerebellar structures. This stochastic behavior suggests that structural alterations alone, while present, may be insufficiently discriminative for effective classification without complementary functional or behavioral indicators.</p>
<p>Conversely, SHapley Additive exPlanations (SHAP) analysis of clinical variables within the optimal multimodal framework revealed hierarchical feature importance with clear mechanistic interpretability (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref>). Sex emerged as the predominant contributor with SHAP values ranging from -0.05 to +0.45, exhibiting pronounced rightward skew indicative of male sex as a risk amplifier. This finding aligns with established sex differences in alcohol metabolism, neurotoxic vulnerability, and addiction trajectories. The prominence of biological sex as a predictor (SHAP values: -0.05 to +0.45) aligns with established epidemiological evidence showing higher AUD prevalence in male firefighters and known sex differences in alcohol metabolism and neurotoxic vulnerability (<xref ref-type="bibr" rid="B55">55</xref>). Non-dominant hand Grooved Pegboard performance demonstrated bidirectional effects (SHAP values: -0.30 to +0.25), suggesting that motor coordination impairments serve as sensitive indicators of subclinical neurological compromise.</p>
<p>The substantial performance differential between the multimodal framework (79.88% accuracy) and both clinical-only (62.53%) and imaging-only (61.53%) approaches cannot be attributed to simple additive effects of feature concatenation. Rather, this performance enhancement reflects fundamental principles of multimodal machine learning wherein complementary information sources capture distinct aspects of underlying pathophysiology (<xref ref-type="bibr" rid="B56">56</xref>, <xref ref-type="bibr" rid="B57">57</xref>). Recent theoretical frameworks in multimodal neuroimaging emphasize that different data modalities provide non-redundant views of complex biological phenomena, with optimal integration strategies exploiting this complementarity to achieve superior discriminative power (<xref ref-type="bibr" rid="B58">58</xref>, <xref ref-type="bibr" rid="B59">59</xref>).</p>
<p>The integration of neuroimaging and clinical features within our multimodal architecture leverages what has been termed &#x201c;cooperative fusion&#x201d; in the multimodal learning literature, wherein modalities interact synergistically to reveal patterns invisible to either modality in isolation (<xref ref-type="bibr" rid="B60">60</xref>). This approach aligns with recent comprehensive reviews demonstrating that transformer and CNN architectures require tailored interpretability methods to effectively capture their distinct feature extraction patterns (<xref ref-type="bibr" rid="B61">61</xref>, <xref ref-type="bibr" rid="B62">62</xref>). Structural MRI captures static morphological alterations reflecting cumulative neurotoxic effects, while neuropsychological assessments provide dynamic functional readouts of neural system integrity. The Vision Transformer component, designed to capture global spatial dependencies, may identify distributed patterns of subtle atrophy that achieve diagnostic relevance only when contextualized by concurrent functional deficits captured through clinical assessments. This synergistic relationship aligns with recent findings demonstrating that multimodal approaches consistently outperform unimodal methods in neuropsychiatric classification tasks by capitalizing on the complementary nature of structural and functional information (<xref ref-type="bibr" rid="B63">63</xref>, <xref ref-type="bibr" rid="B64">64</xref>).</p>
<p>Critically, observed performance gains cannot be explained by overfitting to clinical features or trivial demographic correlations. SHAP analysis reveals that while sex contributes significantly, motor coordination measures and other neuropsychological indicators provide substantial independent predictive value. Moreover, the failure of clinical-only models to exceed 62.53% accuracy demonstrates that behavioral assessments alone lack sufficient discriminative capacity. Similarly, poor performance of imaging-only models indicates that structural alterations, though present, require behavioral anchoring for meaningful interpretation in this at-risk but pre-clinical population.</p>
<p>Mechanistic insights derived from interpretability analyses have profound implications for understanding AUD vulnerability in occupational cohorts. The failure of image-only models to identify consistent neuroanatomical markers suggests that structural alterations in early-stage or at-risk individuals may be subtle, distributed, and heterogeneous, requiring behavioral anchors for meaningful interpretation. The prominence of sex and motor coordination measures in feature importance rankings indicates that integrative models capturing both biological predisposition and functional manifestation provide superior discriminative capacity. These findings support a multifactorial conceptualization of AUD risk wherein neurobiological alterations interact with demographic vulnerabilities and manifest through measurable performance deficits before clinical thresholds are reached.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Methodological limitations</title>
<p>This investigation presents several methodological constraints warranting comprehensive examination. First, the cross-sectional design fundamentally precludes causal inference regarding temporal evolution of structural brain alterations and their relationship to AUD risk. Longitudinal investigations tracking firefighters from recruitment through career progression would be essential to establish whether observed neuroanatomical variations represent predisposing vulnerabilities, consequences of occupational stress exposure, early markers of problematic alcohol use, or complex interactions among these factors. The absence of temporal data particularly limits our ability to determine whether structural alterations precede behavioral manifestations or emerge concurrently with escalating alcohol consumption.</p>
<p>Additionally, exclusive reliance on T1-weighted structural MRI, while strategically chosen for clinical feasibility, imposes inherent constraints on the comprehensiveness of neurobiological characterization. Resting-state functional MRI investigations have identified specific functional connectivity alterations in reward, salience, and executive control networks that differentiate individuals with AUD from controls (<xref ref-type="bibr" rid="B43">43</xref>, <xref ref-type="bibr" rid="B44">44</xref>, <xref ref-type="bibr" rid="B47">47</xref>). Our structural-only approach cannot capture these dynamic network-level disruptions, potentially missing critical neurophysiological markers of addiction vulnerability. Future iterations incorporating abbreviated resting-state protocols or task-based functional MRI targeting reward processing could enhance predictive accuracy while maintaining reasonable clinical practicality. Additionally, advanced structural imaging techniques such as diffusion tensor imaging could provide microstructural integrity measures complementing volumetric assessments.</p>
<p>Furthermore, the computational decision to segment three-dimensional brain volumes into two-dimensional axial slices, while reducing computational complexity and memory requirements, sacrifices spatial continuity information. Three-dimensional convolutional architectures or graph neural networks operating on whole-brain volumes could potentially capture long-range anatomical relationships and improve feature extraction, though at substantially increased computational cost. The trade-off between model sophistication and practical deployability remains a critical consideration for clinical translation.</p>
<p>Moreover, generalizability of our findings faces multiple constraints. The sample comprised exclusively Korean firefighters, introducing both cultural and occupational specificity that may limit applicability to other populations. Cultural variations in alcohol consumption patterns, stigma associated with help-seeking, and occupational stress exposure could influence both structural brain alterations and model performance. The marked sex imbalance (93% male) reflects firefighting workforce demographics but severely limits conclusions about female firefighters, particularly given sex differences in alcohol metabolism, vulnerability to neurotoxic effects, and addiction trajectories. Validation in diverse cultural contexts, occupational groups, and sex-balanced samples remains essential before broader implementation.</p>
<p>Concomitantly, several technical limitations merit consideration. The AUDIT threshold of 8 for defining at-risk status, while internationally validated, may not optimally discriminate problematic drinking patterns in high-functioning occupational cohorts where normative drinking levels differ from general populations. The relatively modest sample size (n=689), while substantial for neuroimaging studies, may limit detection of subtle subgroup differences or complex interaction effects. The absence of genetic data precludes investigation of gene-environment interactions known to influence AUD vulnerability. Missing longitudinal follow-up data prevents validation of the model&#x2019;s actual predictive utility for incident AUD diagnosis or occupational impairment.</p>
<p>Beyond these methodological considerations, clinical implementation faces practical challenges beyond model performance. The requirement for high-resolution structural MRI limits deployment to settings with advanced imaging facilities, potentially excluding rural or resource-limited fire departments. The need for standardized neuropsychological testing by trained personnel adds operational complexity. Privacy concerns regarding neuroimaging-based occupational screening require careful ethical consideration and policy development. The potential for algorithmic bias, particularly given the homogeneous training sample, necessitates ongoing monitoring and recalibration in diverse deployment contexts.</p>
<p>Finally, performance ceiling effects warrant critical consideration. While the achieved 79.88% accuracy represents competitive performance relative to existing multimodal approaches, it nonetheless implies a 20.12% misclassification rate with asymmetric consequences: false positives potentially triggering unwarranted career interventions versus false negatives resulting in missed opportunities for early therapeutic engagement. Furthermore, the absence of longitudinal follow-up data fundamentally constrains clinical interpretation; the model&#x2019;s capacity to predict incident AUD diagnoses, trajectory of symptom progression, or subsequent occupational impairment remains empirically undefined. This temporal limitation restricts current applicability to cross-sectional risk stratification rather than prospective prediction, highlighting the imperative for longitudinal validation studies to establish true predictive validity and optimal screening intervals.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Implications for occupational health screening</title>
<p>The demonstrated feasibility of achieving robust AUD risk classification using structural MRI represents a significant advancement for occupational health surveillance in high-risk professions. Traditional screening paradigms relying on self-report instruments face systematic limitations in emergency response populations where cultural valorization of stoicism, occupational stigma, and career preservation concerns suppress accurate disclosure of alcohol-related problems (<xref ref-type="bibr" rid="B2">2</xref>, <xref ref-type="bibr" rid="B12">12</xref>). The integration of objective neurobiological markers derived from structural neuroimaging circumvents these reporting biases while maintaining classification performance comparable to more complex multimodal approaches.</p>
<p>Implementation within existing occupational health frameworks requires consideration of both technical infrastructure and organizational factors. Fire departments conducting periodic comprehensive medical evaluations could incorporate T1-weighted structural MRI protocols, though the total examination time of 30&#x2013;60 minutes represents a substantial logistical consideration. While the T1-weighted sequence itself requires only 5&#x2013;10 minutes of actual acquisition time, the complete imaging protocol including patient preparation, positioning, and safety procedures necessitates dedicated scheduling within occupational health assessments. The demonstrated predictive value of combining neuroimaging with standardized neuropsychological assessments suggests that brief cognitive testing batteries could enhance screening accuracy without requiring specialized neuropsychological expertise. Automated analysis pipelines utilizing the validated deep learning architecture could provide rapid risk stratification post-acquisition, enabling occupational health physicians to prioritize intervention resources toward highest-risk individuals.</p>
<p>To address feasibility concerns, a tiered implementation strategy could optimize resource utilization while maintaining screening effectiveness. Initial deployment could target high-risk subpopulations identified through traditional screening tools (AUDIT scores &#x2265; 15) or those with documented occupational incidents, thereby concentrating MRI resources on individuals with greatest clinical need. As infrastructure develops and costs decrease, screening criteria could progressively expand to encompass broader firefighter populations. This phased approach aligns with successful precedents in occupational health screening, where targeted protocols for high-risk workers preceded universal implementation (<xref ref-type="bibr" rid="B65">65</xref>).</p>
<p>Critical ethical considerations must guide translation from research findings to occupational screening practices. Clear delineation between probabilistic risk assessment and clinical diagnosis remains essential to prevent discriminatory practices while maximizing preventive potential. Screening protocols should emphasize early intervention and support rather than punitive measures, with explicit protections ensuring that neuroimaging findings cannot adversely impact employment status without corroborating clinical evidence. Longitudinal monitoring frameworks tracking predictive validity of initial risk assessments against subsequent clinical outcomes would enable continuous algorithm refinement while building evidence for screening effectiveness.</p>
<p>Economic implications of neuroimaging-based screening warrant careful analysis within resource allocation frameworks. Recent economic modeling of MRI-based screening programs provides relevant benchmarks. The UK Biobank&#x2019;s population neuroimaging initiative achieved per-scan costs of &#xa3;264 ($330 USD) through high-volume standardization (<xref ref-type="bibr" rid="B66">66</xref>). Considering firefighters&#x2019; elevated AUD risk (56.9% in our cohort vs. 6.2% general population) (<xref ref-type="bibr" rid="B67">67</xref>) and associated costs of untreated AUD (estimated $249 billion annually in the US) (<xref ref-type="bibr" rid="B68">68</xref>), targeted neuroimaging screening may prove cost-effective despite initial infrastructure investments. A threshold analysis suggests that preventing one severe occupational incident per 150 screenings would offset implementation costs. Furthermore, downstream savings from prevented occupational injuries, reduced absenteeism, decreased liability exposure, and maintained operational readiness strengthen the economic justification. Insurance frameworks may require modification to recognize preventive neuroimaging in high-risk occupational cohorts as medically necessary, particularly given these demonstrated cost-benefit ratios. Future development priorities should address current limitations while enhancing clinical utility. Multicenter validation studies incorporating diverse geographical regions, departmental cultures, and demographic compositions would establish generalizability boundaries and identify population-specific calibration requirements. Integration with emerging digital biomarkers from wearable devices, sleep monitoring, and stress physiology could create comprehensive risk profiles extending beyond cross-sectional neuroimaging snapshots. Development of abbreviated screening protocols optimized for rapid deployment during routine medical evaluations could enhance feasibility while maintaining predictive accuracy.</p>
<p>The broader implications extend beyond firefighting to encompass other high-stress occupations with elevated substance use risk, including law enforcement, emergency medical services, and military personnel. Establishing standardized neuroimaging protocols and classification algorithms across these populations would enable comparative effectiveness research while building robust normative databases. International collaboration through occupational health networks could accelerate validation efforts while ensuring equitable access to advanced screening technologies across resource-varied settings.</p>
</sec>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The dataset for this article are not publicly available due to due to ethical and privacy concerns. Requests to access the datasets should be directed to the corresponding author/s.</p>
</sec>
<sec id="s6" sec-type="ethics-statement">
<title>Ethics statement</title>
<p>The studies involving humans were approved by The Institutional Review Board of Ewha Womans University. The studies were conducted in accordance with the local legislation and institutional requirements. The participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>MJ: Writing &#x2013; review &amp; editing, Software, Writing &#x2013; original draft, Methodology, Visualization. DK: Methodology, Visualization, Writing &#x2013; review &amp; editing, Writing &#x2013; original draft, Software. SY: Writing &#x2013; review &amp; editing, Data curation, Conceptualization. HL: Writing &#x2013; review &amp; editing, Supervision, Conceptualization.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. This work was supported by the National Research Foundation of Korea (NRF) grant funded by the Korea government (MSIT)(RS-2024-00457381 and RS-2024-00440371).</p>
</sec>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s12" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fpsyt.2025.1643552/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fpsyt.2025.1643552/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="SupplementaryFile1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cogan</surname> <given-names>N</given-names>
</name>
<name>
<surname>Craig</surname> <given-names>A</given-names>
</name>
<name>
<surname>Milligan</surname> <given-names>L</given-names>
</name>
<name>
<surname>Mccluskey</surname> <given-names>R</given-names>
</name>
<name>
<surname>Burns</surname> <given-names>T</given-names>
</name>
<name>
<surname>Ptak</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>&#x2018;I&#x2019;ve got no PPE to protect my mind&#x2019;: understanding the needs and experiences of first responders exposed to trauma in the workplace</article-title>. <source>Eur J Psychotraumatol</source>. (<year>2024</year>) <volume>15</volume>:<fpage>2395113</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/20008066.2024.2395113</pub-id>, PMID: <pub-id pub-id-type="pmid">39238472</pub-id></citation></ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Haddock</surname> <given-names>CK</given-names>
</name>
<name>
<surname>Jitnarin</surname> <given-names>N</given-names>
</name>
<name>
<surname>Caetano</surname> <given-names>R</given-names>
</name>
<name>
<surname>Jahnke</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Hollerbach</surname> <given-names>BS</given-names>
</name>
<name>
<surname>Kaipust</surname> <given-names>CM</given-names>
</name>
<etal/>
</person-group>. <article-title>Norms about alcohol use among US firefighters</article-title>. <source>Saf Health At Work</source>. (<year>2022</year>) <volume>13</volume>:<page-range>387&#x2013;93</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.shaw.2022.08.008</pub-id>, PMID: <pub-id pub-id-type="pmid">36579011</pub-id></citation></ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yoo</surname> <given-names>JY</given-names>
</name>
<name>
<surname>Sarkar</surname> <given-names>A</given-names>
</name>
<name>
<surname>Song</surname> <given-names>H-S</given-names>
</name>
<name>
<surname>Bang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shim</surname> <given-names>G</given-names>
</name>
<name>
<surname>Springer</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Gut microbiome alterations, mental health, and alcohol consumption: investigating the gut&#x2013;brain axis in firefighters</article-title>. <source>Microorganisms</source>. (<year>2025</year>) <volume>13</volume>:<fpage>680</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/microorganisms13030680</pub-id>, PMID: <pub-id pub-id-type="pmid">40142574</pub-id></citation></ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rotunda</surname> <given-names>RJ</given-names>
</name>
<name>
<surname>Herzog</surname> <given-names>J</given-names>
</name>
<name>
<surname>Dillard</surname> <given-names>DR</given-names>
</name>
<name>
<surname>King</surname> <given-names>E</given-names>
</name>
<name>
<surname>O&#x2019;dare</surname> <given-names>K</given-names>
</name>
</person-group>. <article-title>Alcohol misuse and correlates with mental health indicators among firefighters</article-title>. <source>Subst Use Misuse</source>. (<year>2025</year>) <volume>60</volume>:<page-range>236&#x2013;43</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/10826084.2024.2422975</pub-id>, PMID: <pub-id pub-id-type="pmid">39511710</pub-id></citation></ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Carleton</surname> <given-names>RN</given-names>
</name>
<name>
<surname>Afifi</surname> <given-names>TO</given-names>
</name>
<name>
<surname>Turner</surname> <given-names>S</given-names>
</name>
<name>
<surname>Taillieu</surname> <given-names>T</given-names>
</name>
<name>
<surname>Duranceau</surname> <given-names>S</given-names>
</name>
<name>
<surname>Lebouthillier</surname> <given-names>DM</given-names>
</name>
<etal/>
</person-group>. <article-title>Mental disorder symptoms among public safety personnel in Canada</article-title>. <source>Can J Psychiatry</source>. (<year>2018</year>) <volume>63</volume>:<fpage>54</fpage>&#x2013;<lpage>64</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1177/0706743717723825</pub-id>, PMID: <pub-id pub-id-type="pmid">28845686</pub-id></citation></ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Birrell</surname> <given-names>J</given-names>
</name>
<name>
<surname>Meares</surname> <given-names>K</given-names>
</name>
<name>
<surname>Wilkinson</surname> <given-names>A</given-names>
</name>
<name>
<surname>Freeston</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Toward a definition of intolerance of uncertainty: A review of factor analytical studies of the Intolerance of Uncertainty Scale</article-title>. <source>Clin Psychol Rev</source>. (<year>2011</year>) <volume>31</volume>:<page-range>1198&#x2013;208</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.cpr.2011.07.009</pub-id>, PMID: <pub-id pub-id-type="pmid">21871853</pub-id></citation></ref>
<ref id="B7">
<label>7</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pretorius</surname> <given-names>TB</given-names>
</name>
<name>
<surname>Padmanabhanunni</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>The relationship between intolerance of uncertainty and alcohol use in first responders: A cross-sectional study of the direct, mediating and moderating role of generalized resistance resources</article-title>. <source>Int J Environ Res Public Health</source>. (<year>2025</year>) <volume>22</volume>:<fpage>383</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/ijerph22030383</pub-id>, PMID: <pub-id pub-id-type="pmid">40238397</pub-id></citation></ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hallihan</surname> <given-names>H</given-names>
</name>
<name>
<surname>Bing-Canar</surname> <given-names>H</given-names>
</name>
<name>
<surname>Paltell</surname> <given-names>K</given-names>
</name>
<name>
<surname>Berenz</surname> <given-names>EC</given-names>
</name>
</person-group>. <article-title>Negative urgency, PTSD symptoms, and alcohol risk in college students</article-title>. <source>Addictive Behav Rep</source>. (<year>2023</year>) <volume>17</volume>:<fpage>100480</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.abrep.2023.100480</pub-id>, PMID: <pub-id pub-id-type="pmid">36698484</pub-id></citation></ref>
<ref id="B9">
<label>9</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>M&#xfc;ller</surname> <given-names>CP</given-names>
</name>
<name>
<surname>Schumann</surname> <given-names>G</given-names>
</name>
<name>
<surname>Rehm</surname> <given-names>J</given-names>
</name>
<name>
<surname>Kornhuber</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lenz</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Self-management with alcohol over lifespan: psychological mechanisms, neurobiological underpinnings, and risk assessment</article-title>. <source>Mol Psychiatry</source>. (<year>2023</year>) <volume>28</volume>:<page-range>2683&#x2013;96</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41380-023-02074-3</pub-id>, PMID: <pub-id pub-id-type="pmid">37117460</pub-id></citation></ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ekhtiari</surname> <given-names>H</given-names>
</name>
<name>
<surname>Sangchooli</surname> <given-names>A</given-names>
</name>
<name>
<surname>Carmichael</surname> <given-names>O</given-names>
</name>
<name>
<surname>Moeller</surname> <given-names>FG</given-names>
</name>
<name>
<surname>O&#x2019;donnell</surname> <given-names>P</given-names>
</name>
<name>
<surname>Oquendo</surname> <given-names>MA</given-names>
</name>
<etal/>
</person-group>. <article-title>Neuroimaging biomarkers of addiction</article-title>. <source>Nat Ment Health</source>. (<year>2024</year>) <volume>2</volume>:<page-range>1498&#x2013;517</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s44220-024-00334-x</pub-id>
</citation></ref>
<ref id="B11">
<label>11</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Smith</surname> <given-names>LJ</given-names>
</name>
<name>
<surname>Zegel</surname> <given-names>M</given-names>
</name>
<name>
<surname>Bartlett</surname> <given-names>BA</given-names>
</name>
<name>
<surname>Lebeaut</surname> <given-names>A</given-names>
</name>
<name>
<surname>Vujanovic</surname> <given-names>AA</given-names>
</name>
</person-group>. <article-title>Posttraumatic stress and alcohol use among first responders</article-title>. In: <source>Mental health intervention and treatment of first responders and emergency workers</source>. <publisher-loc>Hershey, PA, USA (IGI Global)</publisher-loc>: <publisher-name>IGI Global Scientific Publishing</publisher-name> (<year>2020</year>).</citation></ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Strudwick</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gayed</surname> <given-names>A</given-names>
</name>
<name>
<surname>Deady</surname> <given-names>M</given-names>
</name>
<name>
<surname>Haffar</surname> <given-names>S</given-names>
</name>
<name>
<surname>Mobbs</surname> <given-names>S</given-names>
</name>
<name>
<surname>Malik</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Workplace mental health screening: a systematic review and meta-analysis</article-title>. <source>Occup Environ Med</source>. (<year>2023</year>) <volume>80</volume>:<page-range>469&#x2013;84</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/oemed-2022-108608</pub-id>, PMID: <pub-id pub-id-type="pmid">37321849</pub-id></citation></ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gonzalez</surname> <given-names>DE</given-names>
</name>
<name>
<surname>Lanham</surname> <given-names>SN</given-names>
</name>
<name>
<surname>Martin</surname> <given-names>SE</given-names>
</name>
<name>
<surname>Cleveland</surname> <given-names>RE</given-names>
</name>
<name>
<surname>Wilson</surname> <given-names>TE</given-names>
</name>
<name>
<surname>Langford</surname> <given-names>EL</given-names>
</name>
<etal/>
</person-group>. <article-title>Firefighter health: A narrative review of occupational threats and countermeasures</article-title>. <source>Healthcare</source>. (<year>2024</year>) <volume>12</volume>:<fpage>440</fpage>. MDPI. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/healthcare12040440</pub-id>, PMID: <pub-id pub-id-type="pmid">38391814</pub-id></citation></ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hurtado</surname> <given-names>M</given-names>
</name>
<name>
<surname>Siefkas</surname> <given-names>A</given-names>
</name>
<name>
<surname>Attwood</surname> <given-names>MM</given-names>
</name>
<name>
<surname>Iqbal</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Hoffman</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Machine learning applications and advancements in alcohol use disorder: A systematic review</article-title>. <source>medRxiv</source>. (<year>2022</year>), <fpage>22276057</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1101/2022.06.06.22276057</pub-id>
</citation></ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kinreich</surname> <given-names>S</given-names>
</name>
<name>
<surname>Meyers</surname> <given-names>JL</given-names>
</name>
<name>
<surname>Maron-Katz</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kamarajan</surname> <given-names>C</given-names>
</name>
<name>
<surname>Pandey</surname> <given-names>AK</given-names>
</name>
<name>
<surname>Chorlian</surname> <given-names>DB</given-names>
</name>
<etal/>
</person-group>. <article-title>Predicting risk for Alcohol Use Disorder using longitudinal data with multimodal biomarkers and family history: a machine learning study</article-title>. <source>Mol Psychiatry</source>. (<year>2021</year>) <volume>26</volume>:<page-range>1133&#x2013;41</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41380-019-0534-x</pub-id>, PMID: <pub-id pub-id-type="pmid">31595034</pub-id></citation></ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smucny</surname> <given-names>J</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>G</given-names>
</name>
<name>
<surname>Davidson</surname> <given-names>I</given-names>
</name>
</person-group>. <article-title>Deep learning in neuroimaging: overcoming challenges with emerging approaches</article-title>. <source>Front Psychiatry</source>. (<year>2022</year>) <volume>13</volume>:<elocation-id>912600</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyt.2022.912600</pub-id>, PMID: <pub-id pub-id-type="pmid">35722548</pub-id></citation></ref>
<ref id="B17">
<label>17</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Caplan</surname> <given-names>B</given-names>
</name>
<name>
<surname>Mendoza</surname> <given-names>JE</given-names>
</name>
</person-group>. <article-title>Edinburgh handedness inventory</article-title>. In: <source>Encyclopedia of clinical neuropsychology</source>. <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2011</year>).</citation></ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Buard</surname> <given-names>I</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Kaizer</surname> <given-names>A</given-names>
</name>
<name>
<surname>Lattanzio</surname> <given-names>L</given-names>
</name>
<name>
<surname>Kluger</surname> <given-names>B</given-names>
</name>
<name>
<surname>Enoka</surname> <given-names>RM</given-names>
</name>
</person-group>. <article-title>Finger dexterity measured by the Grooved Pegboard test indexes Parkinson&#x2019;s motor severity in a tremor-independent manner</article-title>. <source>J Electromyography Kinesiol</source>. (<year>2022</year>) <volume>66</volume>:<fpage>102695</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jelekin.2022.102695</pub-id>, PMID: <pub-id pub-id-type="pmid">36030732</pub-id></citation></ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Reitan</surname> <given-names>RM</given-names>
</name>
</person-group>. <article-title>Validity of the Trail Making Test as an indicator of organic brain damage</article-title>. <source>Perceptual Motor Skills</source>. (<year>1958</year>) <volume>8</volume>:<page-range>271&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2466/pms.1958.8.3.271</pub-id>
</citation></ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Varjacic</surname> <given-names>A</given-names>
</name>
<name>
<surname>Mantini</surname> <given-names>D</given-names>
</name>
<name>
<surname>Demeyere</surname> <given-names>N</given-names>
</name>
<name>
<surname>Gillebert</surname> <given-names>CR</given-names>
</name>
</person-group>. <article-title>Neural signatures of Trail Making Test performance: Evidence from lesion-mapping and neuroimaging studies</article-title>. <source>Neuropsychologia</source>. (<year>2018</year>) <volume>115</volume>:<fpage>78</fpage>&#x2013;<lpage>87</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neuropsychologia.2018.03.031</pub-id>, PMID: <pub-id pub-id-type="pmid">29596856</pub-id></citation></ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Babor</surname> <given-names>TF</given-names>
</name>
<name>
<surname>Higgins-Biddle</surname> <given-names>JC</given-names>
</name>
<name>
<surname>Saunders</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Monteiro</surname> <given-names>MG</given-names>
</name>
</person-group>. <article-title>The alcohol use disorders identification test</article-title>. <publisher-name>World Health Organization</publisher-name>, <publisher-loc>Geneva, Switzerland</publisher-loc> (<year>2001</year>).</citation></ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lundin</surname> <given-names>A</given-names>
</name>
<name>
<surname>Hallgren</surname> <given-names>M</given-names>
</name>
<name>
<surname>Balliu</surname> <given-names>N</given-names>
</name>
<name>
<surname>Forsell</surname> <given-names>Y</given-names>
</name>
</person-group>. <article-title>The use of alcohol use disorders identification test (AUDIT) in detecting alcohol use disorder and risk drinking in the general population: validation of AUDIT using schedules for clinical assessment in neuropsychiatry</article-title>. <source>Alcohol: Clin Exp Res</source>. (<year>2015</year>) <volume>39</volume>:<page-range>158&#x2013;65</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/acer.12593</pub-id>, PMID: <pub-id pub-id-type="pmid">25623414</pub-id></citation></ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jenkinson</surname> <given-names>M</given-names>
</name>
<name>
<surname>Beckmann</surname> <given-names>CF</given-names>
</name>
<name>
<surname>Behrens</surname> <given-names>TE</given-names>
</name>
<name>
<surname>Woolrich</surname> <given-names>MW</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>SM</given-names>
</name>
</person-group>. <article-title>Fsl</article-title>. <source>Neuroimage</source>. (<year>2012</year>) <volume>62</volume>:<page-range>782&#x2013;90</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neuroimage.2011.09.015</pub-id>, PMID: <pub-id pub-id-type="pmid">21979382</pub-id></citation></ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Isensee</surname> <given-names>F</given-names>
</name>
<name>
<surname>Schell</surname> <given-names>M</given-names>
</name>
<name>
<surname>Pflueger</surname> <given-names>I</given-names>
</name>
<name>
<surname>Brugnara</surname> <given-names>G</given-names>
</name>
<name>
<surname>Bonekamp</surname> <given-names>D</given-names>
</name>
<name>
<surname>Neuberger</surname> <given-names>U</given-names>
</name>
<etal/>
</person-group>. <article-title>Automated brain extraction of multisequence MRI using artificial neural networks</article-title>. <source>Hum Brain Mapp</source>. (<year>2019</year>) <volume>40</volume>:<page-range>4952&#x2013;64</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/hbm.24750</pub-id>, PMID: <pub-id pub-id-type="pmid">31403237</pub-id></citation></ref>
<ref id="B25">
<label>25</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rosenbloom</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Pfefferbaum</surname> <given-names>A</given-names>
</name>
</person-group>. <article-title>Magnetic resonance imaging of the living brain: evidence for brain degeneration among alcoholics and recovery with abstinence</article-title>. <source>Alcohol Res Health</source>. (<year>2008</year>) <volume>31</volume>:<fpage>362</fpage>., PMID: <pub-id pub-id-type="pmid">23584010</pub-id></citation></ref>
<ref id="B26">
<label>26</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Park</surname> <given-names>JS</given-names>
</name>
</person-group>. <source>Cross-sectional Atlas of Rhesus Monkey Head: with 0.024-mm pixel size color images</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer Nature</publisher-name> (<year>2022</year>).</citation></ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hao</surname> <given-names>R</given-names>
</name>
<name>
<surname>Namdar</surname> <given-names>K</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L</given-names>
</name>
<name>
<surname>Haider</surname> <given-names>MA</given-names>
</name>
<name>
<surname>Khalvati</surname> <given-names>F</given-names>
</name>
</person-group>. <article-title>A comprehensive study of data augmentation strategies for prostate cancer detection in diffusion-weighted MRI using convolutional neural networks</article-title>. <source>J Digital Imaging</source>. (<year>2021</year>) <volume>34</volume>:<page-range>862&#x2013;76</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10278-021-00478-7</pub-id>, PMID: <pub-id pub-id-type="pmid">34254200</pub-id></citation></ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cossio</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Augmenting medical imaging: a comprehensive catalogue of 65 techniques for enhanced data analysis</article-title>. (<year>2023</year>). arXiv preprint arXiv:2303.01178.</citation></ref>
<ref id="B29">
<label>29</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Abdollahi</surname> <given-names>B</given-names>
</name>
<name>
<surname>Tomita</surname> <given-names>N</given-names>
</name>
<name>
<surname>Hassanpour</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Data augmentation in training deep learning models for medical image analysis</article-title>. In: <source>Deep learners and deep learner descriptors for medical applications</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2020</year>).</citation></ref>
<ref id="B30">
<label>30</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Koonce</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>ResNet 50</article-title>. In: <source>Convolutional neural networks with swift for tensorflow: image recognition and dataset categorization</source>. <publisher-loc>Cham, Switzerland</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2021</year>).</citation></ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dosovitskiy</surname> <given-names>A</given-names>
</name>
<name>
<surname>Beyer</surname> <given-names>L</given-names>
</name>
<name>
<surname>Kolesnikov</surname> <given-names>A</given-names>
</name>
<name>
<surname>Weissenborn</surname> <given-names>D</given-names>
</name>
<name>
<surname>Zhai</surname> <given-names>X</given-names>
</name>
<name>
<surname>Unterthiner</surname> <given-names>T</given-names>
</name>
<etal/>
</person-group>. <article-title>An image is worth 16x16 words: Transformers for image recognition at scale</article-title>. (<year>2020</year>). arXiv preprint arXiv:2010.11929.</citation></ref>
<ref id="B32">
<label>32</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rumelhart</surname> <given-names>DE</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>GE</given-names>
</name>
<name>
<surname>Williams</surname> <given-names>RJ</given-names>
</name>
</person-group>. <article-title>Learning representations by back-propagating errors</article-title>. <source>nature</source>. (<year>1986</year>) <volume>323</volume>:<page-range>533&#x2013;6</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/323533a0</pub-id>
</citation></ref>
<ref id="B33">
<label>33</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arango-Argoty</surname> <given-names>G</given-names>
</name>
<name>
<surname>Kipkogei</surname> <given-names>E</given-names>
</name>
<name>
<surname>Stewart</surname> <given-names>R</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>GJ</given-names>
</name>
<name>
<surname>Patra</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kagiampakis</surname> <given-names>I</given-names>
</name>
<etal/>
</person-group>. <article-title>Pretrained transformers applied to clinical studies improve predictions of treatment efficacy and associated biomarkers</article-title>. <source>Nat Commun</source>. (<year>2025</year>) <volume>16</volume>:<fpage>2101</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-025-57181-2</pub-id>, PMID: <pub-id pub-id-type="pmid">40025003</pub-id></citation></ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>DeLong</surname> <given-names>ER</given-names>
</name>
<name>
<surname>Delong</surname> <given-names>DM</given-names>
</name>
<name>
<surname>Clarke-Pearson</surname> <given-names>DL</given-names>
</name>
</person-group>. <article-title>Comparing the areas under two or more correlated receiver operating characteristic curves: a nonparametric approach</article-title>. <source>Biometrics</source>. (<year>1988</year>) <volume>44</volume>:<page-range>837&#x2013;45</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.2307/2531595</pub-id>
</citation></ref>
<ref id="B35">
<label>35</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Sundararajan</surname> <given-names>M</given-names>
</name>
<name>
<surname>Taly</surname> <given-names>A</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>Q</given-names>
</name>
</person-group>. (<year>2017</year>). <article-title>Axiomatic attribution for deep networks</article-title>, in: <conf-name>Proceedings of the 34th International Conference on Machine Learning (ICML 2017), Proceedings of Machine Learning Research</conf-name>, <publisher-name>PMLR</publisher-name> (online open-access publisher), <volume>70</volume>:<page-range>3319&#x2013;28</page-range>. PMLR. Available online at: <uri xlink:href="http://proceedings.mlr.press/v70/sundararajan17a.html">http://proceedings.mlr.press/v70/sundararajan17a.html</uri>
</citation></ref>
<ref id="B36">
<label>36</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>T</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C</given-names>
</name>
</person-group>. (<year>2016</year>). <article-title>Xgboost: A scalable tree boosting system</article-title>, in: <conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining (KDD '16)</conf-name>, <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery (ACM)</publisher-name>. pp. <page-range>785&#x2013;94</page-range>.</citation></ref>
<ref id="B37">
<label>37</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lundberg</surname> <given-names>SM</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>S-I</given-names>
</name>
</person-group>. <article-title>A unified approach to interpreting model predictions</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2017</year>) <volume>30</volume>:<page-range>4765&#x2013;74</page-range>.</citation></ref>
<ref id="B38">
<label>38</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Selvaraju</surname> <given-names>RR</given-names>
</name>
<name>
<surname>Cogswell</surname> <given-names>M</given-names>
</name>
<name>
<surname>Das</surname> <given-names>A</given-names>
</name>
<name>
<surname>Vedantam</surname> <given-names>R</given-names>
</name>
<name>
<surname>Parikh</surname> <given-names>D</given-names>
</name>
<name>
<surname>Batra</surname> <given-names>D</given-names>
</name>
</person-group>. (<year>2017</year>). <article-title>Grad-cam: Visual explanations from deep networks via gradient-based localization</article-title>, in: <conf-name>Proceedings of the IEEE International Conference on Computer Vision (ICCV 2017)</conf-name>, <publisher-loc>Los Alamitos, CA, USA</publisher-loc>: <publisher-name>IEEE Computer Society</publisher-name>. pp. <page-range>618&#x2013;26</page-range>.</citation></ref>
<ref id="B39">
<label>39</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ribeiro</surname> <given-names>MT</given-names>
</name>
<name>
<surname>Singh</surname> <given-names>S</given-names>
</name>
<name>
<surname>Guestrin</surname> <given-names>C</given-names>
</name>
</person-group>. (<year>2016</year>). <article-title>Why should i trust you?&#x201d; Explaining the predictions of any classifier</article-title>, in: <conf-name>Proceedings of the 22nd ACM SIGKDD International Conference on Knowledge Discovery and Data Mining (KDD '16)</conf-name>, <publisher-loc>New York, NY, USA</publisher-loc>: <publisher-name>Association for Computing Machinery (ACM)</publisher-name>. pp. <page-range>1135&#x2013;44</page-range>.</citation></ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Adebayo</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gilmer</surname> <given-names>J</given-names>
</name>
<name>
<surname>Muelly</surname> <given-names>M</given-names>
</name>
<name>
<surname>Goodfellow</surname> <given-names>I</given-names>
</name>
<name>
<surname>Hardt</surname> <given-names>M</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Sanity checks for saliency maps</article-title>. <source>Adv Neural Inf Process Syst</source>. (<year>2018</year>) <volume>31</volume>:<page-range>9505&#x2013;15</page-range>.</citation></ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zheng</surname> <given-names>G</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>M</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>An attention-based multi-modal MRI fusion model for major depressive disorder diagnosis</article-title>. <source>J Neural Eng</source>. (<year>2023</year>) <volume>20</volume>:<fpage>056037</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1088/1741-2552/ad038c</pub-id>, PMID: <pub-id pub-id-type="pmid">37844568</pub-id></citation></ref>
<ref id="B42">
<label>42</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kanyal</surname> <given-names>A</given-names>
</name>
<name>
<surname>Mazumder</surname> <given-names>B</given-names>
</name>
<name>
<surname>Calhoun</surname> <given-names>VD</given-names>
</name>
<name>
<surname>Preda</surname> <given-names>A</given-names>
</name>
<name>
<surname>Turner</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ford</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Multi-modal deep learning from imaging genomic data for schizophrenia classification</article-title>. <source>Front Psychiatry</source>. (<year>2024</year>) <volume>15</volume>:<elocation-id>1384842</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyt.2024.1384842</pub-id>, PMID: <pub-id pub-id-type="pmid">39006822</pub-id></citation></ref>
<ref id="B43">
<label>43</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Du</surname> <given-names>X</given-names>
</name>
<name>
<surname>Kerich</surname> <given-names>M</given-names>
</name>
<name>
<surname>Lohoff</surname> <given-names>FW</given-names>
</name>
<name>
<surname>Momenan</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Random forest based classification of alcohol dependence patients and healthy controls using resting state MRI</article-title>. <source>Neurosci Lett</source>. (<year>2018</year>) <volume>676</volume>:<fpage>27</fpage>&#x2013;<lpage>33</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.neulet.2018.04.007</pub-id>, PMID: <pub-id pub-id-type="pmid">29626649</pub-id></citation></ref>
<ref id="B44">
<label>44</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vergara</surname> <given-names>VM</given-names>
</name>
<name>
<surname>Espinoza</surname> <given-names>FA</given-names>
</name>
<name>
<surname>Calhoun</surname> <given-names>VD</given-names>
</name>
</person-group>. <article-title>Identifying alcohol use disorder with resting state functional Magnetic Resonance Imaging data: a comparison among machine learning classifiers</article-title>. <source>Front Psychol</source>. (<year>2022</year>) <volume>13</volume>:<elocation-id>867067</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyg.2022.867067</pub-id>, PMID: <pub-id pub-id-type="pmid">35756267</pub-id></citation></ref>
<ref id="B45">
<label>45</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kamarajan</surname> <given-names>C</given-names>
</name>
<name>
<surname>Ardekani</surname> <given-names>BA</given-names>
</name>
<name>
<surname>Pandey</surname> <given-names>AK</given-names>
</name>
<name>
<surname>Kinreich</surname> <given-names>S</given-names>
</name>
<name>
<surname>Pandey</surname> <given-names>G</given-names>
</name>
<name>
<surname>Chorlian</surname> <given-names>DB</given-names>
</name>
<etal/>
</person-group>. <article-title>Random forest classification of alcohol use disorder using fMRI functional connectivity, neuropsychological functioning, and impulsivity measures</article-title>. <source>Brain Sci</source>. (<year>2020</year>) <volume>10</volume>:<fpage>115</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/brainsci10020115</pub-id>, PMID: <pub-id pub-id-type="pmid">32093319</pub-id></citation></ref>
<ref id="B46">
<label>46</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guggenmos</surname> <given-names>M</given-names>
</name>
<name>
<surname>Schmack</surname> <given-names>K</given-names>
</name>
<name>
<surname>Veer</surname> <given-names>IM</given-names>
</name>
<name>
<surname>Lett</surname> <given-names>T</given-names>
</name>
<name>
<surname>Sekutowicz</surname> <given-names>M</given-names>
</name>
<name>
<surname>Sebold</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>A multimodal neuroimaging classifier for alcohol dependence</article-title>. <source>Sci Rep</source>. (<year>2020</year>) <volume>10</volume>:<fpage>298</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-019-56923-9</pub-id>, PMID: <pub-id pub-id-type="pmid">31941972</pub-id></citation></ref>
<ref id="B47">
<label>47</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fede</surname> <given-names>SJ</given-names>
</name>
<name>
<surname>Grodin</surname> <given-names>EN</given-names>
</name>
<name>
<surname>Dean</surname> <given-names>SF</given-names>
</name>
<name>
<surname>Diazgranados</surname> <given-names>N</given-names>
</name>
<name>
<surname>Momenan</surname> <given-names>R</given-names>
</name>
</person-group>. <article-title>Resting state connectivity best predicts alcohol use severity in moderate to heavy alcohol users</article-title>. <source>Neuroimage: Clin</source>. (<year>2019</year>) <volume>22</volume>:<fpage>101782</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.nicl.2019.101782</pub-id>, PMID: <pub-id pub-id-type="pmid">30921611</pub-id></citation></ref>
<ref id="B48">
<label>48</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Park</surname> <given-names>J-H</given-names>
</name>
<name>
<surname>Shin</surname> <given-names>Y-B</given-names>
</name>
<name>
<surname>Jung</surname> <given-names>D</given-names>
</name>
<name>
<surname>Hur</surname> <given-names>J-W</given-names>
</name>
<name>
<surname>Pack</surname> <given-names>SP</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>H-J</given-names>
</name>
<etal/>
</person-group>. <article-title>Machine learning prediction of anxiety symptoms in social anxiety disorder: utilizing multimodal data from virtual reality sessions</article-title>. <source>Front Psychiatry</source>. (<year>2025</year>) <volume>15</volume>:<elocation-id>1504190</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fpsyt.2024.1504190</pub-id>, PMID: <pub-id pub-id-type="pmid">39896993</pub-id></citation></ref>
<ref id="B49">
<label>49</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rapisarda</surname> <given-names>F</given-names>
</name>
<name>
<surname>Lanovaz</surname> <given-names>MJ</given-names>
</name>
<name>
<surname>Guay</surname> <given-names>S</given-names>
</name>
<name>
<surname>Geoffrion</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Machine learning models to predict posttraumatic stress injuries in a sample of firefighters: A proof of concept</article-title>. <source>Int J Ment Health</source>. (<year>2025</year>), <fpage>1</fpage>&#x2013;<lpage>21</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1080/00207411.2025.2486084</pub-id>
</citation></ref>
<ref id="B50">
<label>50</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shinde</surname> <given-names>SS</given-names>
</name>
<name>
<surname>Ghotkar</surname> <given-names>AS</given-names>
</name>
</person-group>. <article-title>Mental stress detection with the multimodal data using ensemble optimization enabled explainable convolutional neural network</article-title>. <source>Biomed Mater Devices</source>. (<year>2025</year>), <fpage>1</fpage>&#x2013;<lpage>23</lpage>.</citation></ref>
<ref id="B51">
<label>51</label>
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Chung</surname> <given-names>M</given-names>
</name>
<name>
<surname>Won</surname> <given-names>JB</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>G</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Ozbulak</surname> <given-names>U</given-names>
</name>
</person-group>. (<year>2024</year>). <article-title>Evaluating visual explanations of attention maps for transformer-based medical imaging</article-title>, in: <conf-name>International Conference on Medical Image Computing and Computer-Assisted Intervention</conf-name>. <volume>15011</volume>:<page-range>110&#x2013;20</page-range>. <publisher-name>Springer</publisher-name>.</citation></ref>
<ref id="B52">
<label>52</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fritz</surname> <given-names>M</given-names>
</name>
<name>
<surname>Klawonn</surname> <given-names>AM</given-names>
</name>
<name>
<surname>Zahr</surname> <given-names>NM</given-names>
</name>
</person-group>. <article-title>Neuroimaging in alcohol use disorder: From mouse to man</article-title>. <source>J Neurosci Res</source>. (<year>2022</year>) <volume>100</volume>:<page-range>1140&#x2013;58</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/jnr.24423</pub-id>, PMID: <pub-id pub-id-type="pmid">31006907</pub-id></citation></ref>
<ref id="B53">
<label>53</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>K</given-names>
</name>
<name>
<surname>Huo</surname> <given-names>J</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>D</given-names>
</name>
<etal/>
</person-group>. <article-title>Explainable and interpretable multimodal large language models: A comprehensive survey</article-title>. (<year>2024</year>). arXiv preprint arXiv:2412.02104.</citation></ref>
<ref id="B54">
<label>54</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Anomaly detection in medical via multimodal foundation models</article-title>. <source>Front Bioengineering Biotechnol</source>. (<year>2025</year>) <volume>13</volume>:<elocation-id>1644697</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fbioe.2025.1644697</pub-id>, PMID: <pub-id pub-id-type="pmid">40873433</pub-id></citation></ref>
<ref id="B55">
<label>55</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fama</surname> <given-names>R</given-names>
</name>
<name>
<surname>Le Berre</surname> <given-names>A-P</given-names>
</name>
<name>
<surname>Sassoon</surname> <given-names>SA</given-names>
</name>
<name>
<surname>Zahr</surname> <given-names>NM</given-names>
</name>
<name>
<surname>Pohl</surname> <given-names>KM</given-names>
</name>
<name>
<surname>Pfefferbaum</surname> <given-names>A</given-names>
</name>
<etal/>
</person-group>. <article-title>Relations between cognitive and motor deficits and regional brain volumes in individuals with alcoholism</article-title>. <source>Brain Structure Funct</source>. (<year>2019</year>) <volume>224</volume>:<page-range>2087&#x2013;101</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s00429-019-01894-w</pub-id>, PMID: <pub-id pub-id-type="pmid">31161472</pub-id></citation></ref>
<ref id="B56">
<label>56</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Tulay</surname> <given-names>EE</given-names>
</name>
<name>
<surname>Metin</surname> <given-names>B</given-names>
</name>
<name>
<surname>Tarhan</surname> <given-names>N</given-names>
</name>
<name>
<surname>Arikan</surname> <given-names>MK</given-names>
</name>
</person-group>. <article-title>Multimodal neuroimaging: basic concepts and classification of neuropsychiatric diseases</article-title>. <source>Clin EEG Neurosci</source>. (<year>2019</year>) <volume>50</volume>:<fpage>20</fpage>&#x2013;<lpage>33</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1177/1550059418782093</pub-id>, PMID: <pub-id pub-id-type="pmid">29925268</pub-id></citation></ref>
<ref id="B57">
<label>57</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Stahlschmidt</surname> <given-names>SR</given-names>
</name>
<name>
<surname>Ulfenborg</surname> <given-names>B</given-names>
</name>
<name>
<surname>Synnergren</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Multimodal deep learning for biomedical data fusion: a review</article-title>. <source>Briefings Bioinf</source>. (<year>2022</year>) <volume>23</volume>:<fpage>bbab569</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/bib/bbab569</pub-id>, PMID: <pub-id pub-id-type="pmid">35089332</pub-id></citation></ref>
<ref id="B58">
<label>58</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Calhoun</surname> <given-names>VD</given-names>
</name>
<name>
<surname>Sui</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Multimodal fusion of brain imaging data: a key to finding the missing link (s) in complex mental illness</article-title>. <source>Biol Psychiatry: Cogn Neurosci Neuroimaging</source>. (<year>2016</year>) <volume>1</volume>:<page-range>230&#x2013;44</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.bpsc.2015.12.005</pub-id>, PMID: <pub-id pub-id-type="pmid">27347565</pub-id></citation></ref>
<ref id="B59">
<label>59</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y-D</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S-H</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Yao</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Q</given-names>
</name>
<etal/>
</person-group>. <article-title>Advances in multimodal data fusion in neuroimaging: Overview, challenges, and novel orientation</article-title>. <source>Inf Fusion</source>. (<year>2020</year>) <volume>64</volume>:<page-range>149&#x2013;87</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.inffus.2020.07.006</pub-id>, PMID: <pub-id pub-id-type="pmid">32834795</pub-id></citation></ref>
<ref id="B60">
<label>60</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Baltru&#x161;aitis</surname> <given-names>T</given-names>
</name>
<name>
<surname>Ahuja</surname> <given-names>C</given-names>
</name>
<name>
<surname>Morency</surname> <given-names>L-P</given-names>
</name>
</person-group>. <article-title>Multimodal machine learning: A survey and taxonomy</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source>. (<year>2018</year>) <volume>41</volume>:<page-range>423&#x2013;43</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2018.2798607</pub-id>, PMID: <pub-id pub-id-type="pmid">29994351</pub-id></citation></ref>
<ref id="B61">
<label>61</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pu</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Xi</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>S</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>L</given-names>
</name>
</person-group>. <article-title>Advantages of transformer and its application for medical image segmentation: a survey</article-title>. <source>Biomed Eng Online</source>. (<year>2024</year>) <volume>23</volume>:<fpage>14</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1186/s12938-024-01212-4</pub-id>, PMID: <pub-id pub-id-type="pmid">38310297</pub-id></citation></ref>
<ref id="B62">
<label>62</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Azad</surname> <given-names>R</given-names>
</name>
<name>
<surname>Kazerouni</surname> <given-names>A</given-names>
</name>
<name>
<surname>Heidari</surname> <given-names>M</given-names>
</name>
<name>
<surname>Aghdam</surname> <given-names>EK</given-names>
</name>
<name>
<surname>Molaei</surname> <given-names>A</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Advances in medical image analysis with vision transformers: a comprehensive review</article-title>. <source>Med Image Anal</source>. (<year>2024</year>) <volume>91</volume>:<fpage>103000</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.media.2023.103000</pub-id>, PMID: <pub-id pub-id-type="pmid">37883822</pub-id></citation></ref>
<ref id="B63">
<label>63</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Salvador</surname> <given-names>R</given-names>
</name>
<name>
<surname>Radua</surname> <given-names>J</given-names>
</name>
<name>
<surname>Canales-Rodr&#xed;guez</surname> <given-names>EJ</given-names>
</name>
<name>
<surname>Solanes</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sarr&#xf3;</surname> <given-names>S</given-names>
</name>
<name>
<surname>Goikolea</surname> <given-names>JM</given-names>
</name>
<etal/>
</person-group>. <article-title>Evaluation of machine learning algorithms and structural features for optimal MRI-based diagnostic prediction in psychosis</article-title>. <source>PloS One</source>. (<year>2017</year>) <volume>12</volume>:<fpage>e0175683</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0175683</pub-id>, PMID: <pub-id pub-id-type="pmid">28426817</pub-id></citation></ref>
<ref id="B64">
<label>64</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sui</surname> <given-names>J</given-names>
</name>
<name>
<surname>Adali</surname> <given-names>T</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J</given-names>
</name>
<name>
<surname>Calhoun</surname> <given-names>VD</given-names>
</name>
</person-group>. <article-title>A review of multivariate methods for multimodal fusion of brain imaging data</article-title>. <source>J Neurosci Methods</source>. (<year>2012</year>) <volume>204</volume>:<fpage>68</fpage>&#x2013;<lpage>81</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.jneumeth.2011.10.031</pub-id>, PMID: <pub-id pub-id-type="pmid">22108139</pub-id></citation></ref>
<ref id="B65">
<label>65</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Koh</surname> <given-names>D</given-names>
</name>
<name>
<surname>Aw</surname> <given-names>T-C</given-names>
</name>
</person-group>. <article-title>Surveillance in occupational health</article-title>. <source>Occup Environ Med</source>. (<year>2003</year>) <volume>60</volume>:<page-range>705&#x2013;10</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1136/oem.60.9.705</pub-id>, PMID: <pub-id pub-id-type="pmid">12937199</pub-id></citation></ref>
<ref id="B66">
<label>66</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Littlejohns</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Holliday</surname> <given-names>J</given-names>
</name>
<name>
<surname>Gibson</surname> <given-names>LM</given-names>
</name>
<name>
<surname>Garratt</surname> <given-names>S</given-names>
</name>
<name>
<surname>Oesingmann</surname> <given-names>N</given-names>
</name>
<name>
<surname>Alfaro-Almagro</surname> <given-names>F</given-names>
</name>
<etal/>
</person-group>. <article-title>The UK Biobank imaging enhancement of 100,000 participants: rationale, data collection, management and future directions</article-title>. <source>Nat Commun</source>. (<year>2020</year>) <volume>11</volume>:<fpage>2624</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-020-15948-9</pub-id>, PMID: <pub-id pub-id-type="pmid">32457287</pub-id></citation></ref>
<ref id="B67">
<label>67</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Grant</surname> <given-names>BF</given-names>
</name>
<name>
<surname>Goldstein</surname> <given-names>RB</given-names>
</name>
<name>
<surname>Saha</surname> <given-names>TD</given-names>
</name>
<name>
<surname>Chou</surname> <given-names>SP</given-names>
</name>
<name>
<surname>Jung</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Epidemiology of DSM-5 alcohol use disorder: results from the National Epidemiologic Survey on Alcohol and Related Conditions III</article-title>. <source>JAMA Psychiatry</source>. (<year>2015</year>) <volume>72</volume>:<page-range>757&#x2013;66</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1001/jamapsychiatry.2015.0584</pub-id>, PMID: <pub-id pub-id-type="pmid">26039070</pub-id></citation></ref>
<ref id="B68">
<label>68</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sacks</surname> <given-names>JJ</given-names>
</name>
<name>
<surname>Gonzales</surname> <given-names>KR</given-names>
</name>
<name>
<surname>Bouchery</surname> <given-names>EE</given-names>
</name>
<name>
<surname>Tomedi</surname> <given-names>LE</given-names>
</name>
<name>
<surname>Brewer</surname> <given-names>RD</given-names>
</name>
</person-group>. <article-title>2010 national and state costs of excessive alcohol consumption</article-title>. <source>Am J Prev Med</source>. (<year>2015</year>) <volume>49</volume>:<page-range>e73&#x2013;9</page-range>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.amepre.2015.05.031</pub-id>, PMID: <pub-id pub-id-type="pmid">26477807</pub-id></citation></ref>
</ref-list>
</back>
</article>