<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3-mathml3.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" dtd-version="1.3" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title-group>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
</journal-title-group>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2025.1621244</article-id>
<article-version article-version-type="Version of Record" vocab="NISO-RP-8-2008"/>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Original Research</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Robust automated preclinical fMRI preprocessing via a multi-stage dilated convolutional Swin Transformer affine registration</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Soltanpour</surname> <given-names>Sima</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Methodology" vocab-term-identifier="https://credit.niso.org/contributor-roles/methodology/">Methodology</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/2953624"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Nasseef</surname> <given-names>Md. Taufiq</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Validation" vocab-term-identifier="https://credit.niso.org/contributor-roles/validation/">Validation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Visualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/visualization/">Visualization</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing &#x2013; original draft</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/362660"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Utama</surname> <given-names>Rachel</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Formal analysis" vocab-term-identifier="https://credit.niso.org/contributor-roles/formal-analysis/">Formal analysis</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Investigation" vocab-term-identifier="https://credit.niso.org/contributor-roles/investigation/">Investigation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Chang</surname> <given-names>Arnold</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Software" vocab-term-identifier="https://credit.niso.org/contributor-roles/software/">Software</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Madularu</surname> <given-names>Dan</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
<contrib contrib-type="author">
<name><surname>Kulkarni</surname> <given-names>Praveen</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Data curation" vocab-term-identifier="https://credit.niso.org/contributor-roles/data-curation/">Data curation</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/203172"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ferris</surname> <given-names>Craig F.</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
<uri xlink:href="https://loop.frontiersin.org/people/130043"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Joslin</surname> <given-names>Chris</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Funding acquisition" vocab-term-identifier="https://credit.niso.org/contributor-roles/funding-acquisition/">Funding acquisition</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Project administration" vocab-term-identifier="https://credit.niso.org/contributor-roles/project-administration/">Project administration</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Resources" vocab-term-identifier="https://credit.niso.org/contributor-roles/resources/">Resources</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Supervision" vocab-term-identifier="https://credit.niso.org/contributor-roles/supervision/">Supervision</role>
<role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing &#x2013; review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing &#x2013; review &#x00026; editing</role>
</contrib>
</contrib-group>
<aff id="aff1"><label>1</label><institution>School of Information Technology, Carleton University</institution>, <city>Ottawa, ON</city>, <country country="ca">Canada</country></aff>
<aff id="aff2"><label>2</label><institution>Department of Mathematics, College of Science and Humanity Studies, Prince Sattam Bin Abdulaziz University</institution>, <city>Riyadh</city>, <country country="sa">Saudi Arabia</country></aff>
<aff id="aff3"><label>3</label><institution>Center for Translational NeuroImaging (CTNI), Northeastern University</institution>, <city>Boston, MA</city>, <country country="us">United States</country></aff>
<aff id="aff4"><label>4</label><institution>Tessellis Ltd.</institution>, <city>Ottawa, ON</city>, <country country="ca">Canada</country></aff>
<author-notes>
<corresp id="c001"><label>&#x0002A;</label>Correspondence: Sima Soltanpour, <email xlink:href="mailto:simasoltanpour@cunet.carleton.ca">simasoltanpour@cunet.carleton.ca</email></corresp>
</author-notes>
<pub-date publication-format="electronic" date-type="pub" iso-8601-date="2025-12-12">
<day>12</day>
<month>12</month>
<year>2025</year>
</pub-date>
<pub-date publication-format="electronic" date-type="collection">
<year>2025</year>
</pub-date>
<volume>19</volume>
<elocation-id>1621244</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="rev-recd">
<day>18</day>
<month>11</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>11</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Soltanpour, Nasseef, Utama, Chang, Madularu, Kulkarni, Ferris and Joslin.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Soltanpour, Nasseef, Utama, Chang, Madularu, Kulkarni, Ferris and Joslin</copyright-holder>
<license>
<ali:license_ref start_date="2025-12-12">https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This is an open-access article distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License (CC BY)</ext-link>. The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</license-p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Accurate preprocessing of functional magnetic resonance imaging (fMRI) data is crucial for effective analysis in preclinical studies. Key steps such as denoising, skull-stripping, and affine registration are essential to align fMRI data with a standard atlas. However, challenges such as low resolution, variations in brain geometry, and limited dataset sizes often hinder the performance of traditional and deep learning-based methods.</p></sec>
<sec>
<title>Methods</title>
<p>To address these challenges, we propose a preclinical fMRI preprocessing pipeline that integrates advanced deep learning modules, with a particular focus on a newly developed Swin Transformer-based affine registration method. The pipeline incorporates our previously established modules for 3D Generative Adversarial Network (GAN)-based denoising and Transformer-based skull stripping, followed by the proposed Multi-stage Dilated Convolutional Swin Transformer (MsDCSwinT) for affine registration. This new registration method captures both local and global spatial misalignments, ensuring accurate alignment with a standard atlas even in challenging preclinical datasets.</p></sec>
<sec>
<title>Results</title>
<p>We validate the pipeline across multiple preclinical fMRI studies and demonstrate that our affine registration module achieves higher average Dice similarity coefficients compared to state-of-the-art methods.</p></sec>
<sec>
<title>Discussion</title>
<p>By leveraging GANs and Transformers, our pipeline offers a robust, accurate, and fully automated solution for preclinical fMRI.</p></sec></abstract>
<kwd-group>
<kwd>functional MRI</kwd>
<kwd>preprocessing pipeline</kwd>
<kwd>affine registration</kwd>
<kwd>transformers</kwd>
<kwd>deep learning</kwd>
</kwd-group>
<funding-group>
<award-group id="gs1">
<funding-source id="sp1">
<institution-wrap>
<institution>Mitacs</institution>
<institution-id institution-id-type="doi" vocab="open-funder-registry" vocab-identifier="10.13039/open_funder_registry">10.13039/501100004489</institution-id>
</institution-wrap>
</funding-source>
</award-group>
<funding-statement>The author(s) declare that financial support was received for the research and/or publication of this article. This study was funded by Mitacs Accelerate grant (grant number: IT40950).</funding-statement>
</funding-group>
<counts>
<fig-count count="7"/>
<table-count count="2"/>
<equation-count count="20"/>
<ref-count count="47"/>
<page-count count="17"/>
<word-count count="11360"/>
</counts>
<custom-meta-group>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Neural Technology</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<label>1</label>
<title>Introduction</title>
<p>Small animal models play a crucial role in preclinical research, aiding in the evaluation of new pharmaceutical compounds and the investigation of biological functions (<xref ref-type="bibr" rid="B19">Hasani et al., 2025</xref>). Rats are one of the most commonly used species in medical studies due to their compact size, rapid reproduction, genetic resemblance to humans, and their ability to model various human diseases (<xref ref-type="bibr" rid="B45">Szpirer, 2020</xref>). Advancements in imaging technologies enable the continuous, non-invasive examination of both anatomical structures and biological processes in small animals. Functional magnetic resonance imaging (fMRI) is a non-invasive imaging tool used to assess brain structure and function, which works with magnetic fields and radiofrequency pulses (<xref ref-type="bibr" rid="B40">Ren et al., 2022</xref>).</p>
<p>fMRI preprocessing is a critical step in neuroimaging analysis, ensuring that raw fMRI data is corrected for artifacts, aligned to a standard space, and optimized for subsequent statistical and machine learning-based analyses. A typical fMRI preprocessing pipeline includes essential steps such as denoising, motion correction, skull-stripping, and registration to an atlas. These steps are crucial for minimizing variability across scans, improving signal quality, and enabling accurate group-level comparisons (<xref ref-type="bibr" rid="B39">Nieto-Castanon, 2022</xref>; <xref ref-type="bibr" rid="B10">Di and Biswal, 2023</xref>). However, traditional preprocessing methods, including conventional optimization-based registration and CNN-based approaches, often struggle with low-resolution fMRI data, anatomical variations, and limited sample sizes, particularly in preclinical studies.</p>
<p>Rigid and affine registration play a key role in medical imaging and have been widely studied for years. Deep learning has demonstrated advancements in medical image registration. Some existing approaches rely on supervised learning frameworks, which require extensive annotated datasets and domain-specific expertise (<xref ref-type="bibr" rid="B6">Chen et al., 2021</xref>, <xref ref-type="bibr" rid="B7">2022</xref>). Supervised learning approaches for image registration depend heavily on accurately labelled data, which poses a significant challenge in fMRI analysis due to the variability in manually segmented maps across different brain regions. As a result, such methods may be impractical in scenarios where ground-truth registration labels are unavailable, as is the case in this study. To address this limitation, unsupervised and weakly supervised learning strategies have gained attention as viable alternatives, reducing the reliance on precise manual annotations while still enabling effective image alignment (<xref ref-type="bibr" rid="B32">Mok and Chung, 2022</xref>; <xref ref-type="bibr" rid="B27">Ji and Yang, 2024</xref>; <xref ref-type="bibr" rid="B17">Golestani et al., 2025</xref>). Although these approaches offer potential advantages over supervised methods, their applicability to preclinical fMRI image registration remains unexplored and requires further research (<xref ref-type="bibr" rid="B6">Chen et al., 2021</xref>). Moreover, weakly supervised registration methods rely on segmentation labels as supervision, making their performance highly sensitive to segmentation variability. In fMRI datasets, where the number and shape of segmented regions can vary significantly across subjects, this dependence further degrades consistency and accuracy. Therefore, adopting an unsupervised learning strategy is more appropriate for achieving robust registration in preclinical fMRI analysis.</p>
<p>In many image registration systems, images are first aligned using rigid or affine transformations before applying non-rigid or deformable methods (<xref ref-type="bibr" rid="B9">De Vos et al., 2019</xref>). This step helps correct large-scale misalignments between images (<xref ref-type="bibr" rid="B32">Mok and Chung, 2022</xref>). Recent learning-based methods for deformable image registration rely heavily on accurate affine alignment using traditional techniques (<xref ref-type="bibr" rid="B31">Mok and Chung, 2020a</xref>, <xref ref-type="bibr" rid="B34">2021</xref>). While these conventional methods provide high registration accuracy, they can be slow, especially for 4D fMRI images, as processing time depends on the level of misalignment. To enable faster, automated registration, some studies have explored using convolutional neural networks (CNNs) to learn both affine and deformable registration together (<xref ref-type="bibr" rid="B20">Huang et al., 2021</xref>; <xref ref-type="bibr" rid="B22">Iglesias, 2023</xref>). Many of these studies concentrate mainly on enhancing deformable registration, often treating affine registration as a basic preliminary step or neglecting it altogether (<xref ref-type="bibr" rid="B17">Golestani et al., 2025</xref>). As a result, the independent performance of the affine subnetwork, in comparison to existing affine registration techniques, remains unexplored. Since affine transformation deals with global alignment and large displacements, CNNs may not be the best choice for capturing image orientation and absolute position in Cartesian space (<xref ref-type="bibr" rid="B32">Mok and Chung, 2022</xref>).</p>
<p>In human brain imaging, registration benefits from specialized high-level functions such as the FSL package (<xref ref-type="bibr" rid="B26">Jenkinson et al., 2012</xref>) and the ANTS package (<xref ref-type="bibr" rid="B2">Avants et al., 2011</xref>), which are optimized for the size and spatial characteristics of the human brain. In contrast, small animal brain imaging often relies on these same high-level functions, adapting the data to fit the function parameters rather than tailoring the functions to the data. This approach can compromise data accuracy, restrict optimization possibilities, and pose significant challenges to advancing methodologies in small animal brain imaging.</p>
<p>While substantial progress has been made in developing robust, general-purpose preprocessing tools for human imaging data (<xref ref-type="bibr" rid="B13">Esteban et al., 2019</xref>), the preclinical field still lacks similarly reliable and standardized solutions. An automatic PET/MRI registration for preclinical studies based on B-splines and non-linear intensity transformation has been proposed by <xref ref-type="bibr" rid="B4">Bricq et al. (2018)</xref>. A preclinical registration framework was introduced by <xref ref-type="bibr" rid="B1">Anderson et al. (2019)</xref> to address structural imaging, particularly voxel-based morphometry (VBM). A registration workflow for small animal brain MRI is proposed by <xref ref-type="bibr" rid="B23">Ioanas et al. (2021)</xref>. However, comprehensive preprocessing pipelines that effectively handle functional preclinical data remain limited. While prior studies have proposed preclinical imaging workflows (<xref ref-type="bibr" rid="B1">Anderson et al., 2019</xref>; <xref ref-type="bibr" rid="B23">Ioanas et al., 2021</xref>), they primarily focus on structural MRI registration and analysis. In contrast, to the best of our knowledge, our proposed pipeline is the first preprocessing pipeline which integrates deep learning-based modules specifically designed for functional preclinical MRI data.</p>
<p>In this paper, we introduce a novel preclinical fMRI preprocessing pipeline that integrates advanced deep learning techniques, including Swin Transformer-based registration. Our pipeline incorporates our recently developed GAN-based denoising method (<xref ref-type="bibr" rid="B41">Soltanpour et al., 2025a</xref>), which effectively reduces noise while preserving critical functional details, and our transformer-based skull-stripping approach (<xref ref-type="bibr" rid="B42">Soltanpour et al., 2025b</xref>), which ensures precise brain extraction. Additionally, we propose a new affine registration method leveraging Swin Transformer (<xref ref-type="bibr" rid="B29">Liu et al., 2021</xref>) and dilated convolutional block (<xref ref-type="bibr" rid="B16">Gao et al., 2022</xref>), enabling more accurate and efficient alignment of preclinical functional MRI data. By combining these techniques, our pipeline addresses key challenges in preclinical imaging, improving data quality and facilitating more reliable downstream analyses. The general framework of the proposed pipeline is illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. The preprocessing pipeline contains four modules including our GAN-based denoising, motion correction using AFNI (<xref ref-type="bibr" rid="B8">Cox, 1996</xref>), our transformer-based skull stripping, and the proposed Multi-stage Dilated Convolutinal Swin Transformer (MsDCSwinT) affine registration.</p>
<fig position="float" id="F1">
<label>Figure 1</label>
<caption><p>Overview of the proposed deep learning-based preclinical fMRI preprocessing pipeline. The framework integrates four key modules: (1) GAN-based denoising to suppress noise while preserving functional details, (2) motion correction using AFNI, (3) transformer-based skull stripping for accurate brain extraction, and (4) affine registration using the proposed Multi-stage Dilated Convolutional Swin Transformer (MsDCSwinT) for precise spatial alignment.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1621244-g0001.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a process for 3D U-WGAN denoising and SST-DUNet skull stripping of fMRI data. Features include patch extraction and merging, use of a Dense UNet, proposed affine registration, and AFNI motion correction. Multiple steps involve visual representations of fMRI and atlas data.</alt-text>
</graphic>
</fig>
<p>The main contributions of this work are threefold. First, we present a novel end-to-end preprocessing pipeline tailored for preclinical fMRI data, which combines advanced deep learning methods to address the unique challenges of small animal brain imaging. Second, we introduce a GAN-based denoising method that effectively suppresses noise while preserving functional signal integrity, and a transformer-based skull stripping approach that ensures accurate and consistent brain extraction across subjects. Third, we propose a new affine registration framework, Multi-stage Dilated Convolutional Swin Transformer (MsDCSwinT), that improves the precision and speed of spatial alignment in preclinical fMRI datasets by leveraging hierarchical attention mechanisms and dilated convolutions. To the best of our knowledge, this is the first work to introduce a deep learning-based preprocessing pipeline specifically designed for preclinical fMRI studies. Together, these contributions offer a comprehensive and automated solution that enhances the robustness, accuracy, and efficiency of preclinical functional neuroimaging workflows.</p>
<p>The remainder of the paper is organized as follows: Section 2 presents the materials used in this study and introduces a novel affine registration algorithm based on a multi-stage Swin Transformer architecture. Section 3 reports the experimental results. Section 4 discusses the findings and outlines the study&#x00027;s limitations. Finally, Section 5 concludes the paper and outlines directions for future work.</p></sec>
<sec sec-type="materials|methods" id="s2">
<label>2</label>
<title>Materials and methods</title>
<sec>
<label>2.1</label>
<title>Datasets</title>
<p>This study utilized two in-house datasets comprising seven studies and a total of 280 rats, collected by the Center for Translational NeuroImaging (CTNI) at Northeastern University, Boston, MA, USA. No new animals were scanned for this work. All imaging was previously performed using a Bruker Biospec 7.0T/20-cm USR horizontal magnet with a 20-G/cm gradient insert (ID = 12 cm, 120-&#x003BC;s rise time), and data were acquired using built-in quad-coil electronics within the animal restrainer. Male Sprague Dawley rats (325&#x02013;350 g) from Charles River Laboratories (Wilmington, MA, USA) were housed under standard 12:12 h light-dark conditions with unrestricted access to food and water. All animal procedures followed the <italic>Guide for the Care and Use of Laboratory Animals</italic> (NIH Publication No. 85-23, Revised 1985) and were approved by the Institutional Animal Care and Use Committee at Northeastern University, adhering to NIH and AALAS guidelines.</p>
<p>For each imaging session, high-resolution anatomical scans were acquired using a RARE sequence (25 slices, 1 mm thickness, FOV 3.0 cm, resolution 256 &#x000D7; 256, TR = 2.5 s, TE = 12.4 ms, NEX = 6, &#x0007E;6 min total). Task-based fMRI data were collected using a Half-Fourier, single-shot turbo-spin echo (RARE-st) sequence with 96 &#x000D7; 96 in-plane resolution, 20&#x02013;25 slices, TR = 6,000 ms, TE = 48 ms, RARE factor = 36, NEX = 1, repeated 100 times over a 10-min session. Resting-state fMRI (rsfMRI) data were acquired before and after the task-based scans using a spin-echo triple-shot EPI sequence (96 &#x000D7; 96 &#x000D7; [20&#x02013;25 slices], voxel size = 0.312 &#x000D7; 0.312 mm, slice thickness = 1.2 mm, TR = 1000 ms, TE = 15 ms, 300 repetitions, &#x0007E;15 min total scan time).</p>
<sec>
<label>2.1.1</label>
<title>Data for training and test</title>
<p>We apply the proposed affine model for registering rat brain functional images. We applied 6 studies (dataset 1) containing 270 samples for training and test. To train the model, we initially created a training dataset by randomly selecting 80% of the data, reserving the remaining 20% for final performance testing. During the training phase, an additional 80% of the data were randomly sampled from the training dataset, leaving the remaining 20% for validating the training of the model. This training-validation process was repeated five times to ensure an unbiased data distribution. Subsequently, the model with the highest averaged validation accuracy was chosen as the final model for testing. To evaluate the model&#x00027;s generalization ability, we applied one study (dataset 2), containing 10 subjects which has not been applied for training.</p></sec>
<sec>
<label>2.1.2</label>
<title>Data preprocessing for affine registration</title>
<p>Data preprocessing before affine registration is critical for enhancing the model&#x00027;s robustness and generalization. We apply a standard preprocessing steps, including denoising, motion correction, and skull stripping. In our affine registration method, we align the fMRI scans to a standardized 256 &#x000D7; 256 &#x000D7; 64 atlas (<xref ref-type="bibr" rid="B46">Wang et al., 2025</xref>; <xref ref-type="bibr" rid="B15">Fuini et al., 2025</xref>). This atlas, developed by Ekam Imaging (Boston, MA, USA), provides a consistent anatomical reference that facilitates accurate spatial normalization across subjects. By registering all scans to this common space, we ensure comparability in subsequent analysis steps, enabling robust group-level statistical studies and downstream applications in functional neuroimaging. It should be noted that this step performs only affine (global) alignment rather than deformable (non-linear) registration. The proposed MsDCSwinT model estimates affine parameters (translation, rotation, scaling, and shearing) to bring each subject&#x00027;s fMRI volume into coarse correspondence with the standardized atlas space. This affine transformation provides a globally aligned reference frame for downstream group analyses. Deformable or non-linear alignment, which refines local anatomical correspondence, is beyond the current scope and remains part of our planned future work.</p>
<p>fMRI data are inherently noisy due to physiological, hardware-related, and external artifacts, making denoising a critical preprocessing step. As shown in <xref ref-type="fig" rid="F1">Figure 1</xref>, we apply our 3D U-WGAN (<xref ref-type="bibr" rid="B41">Soltanpour et al., 2025a</xref>), a structure-preserving denoising method based on a 3D Wasserstein GAN with a 3D dense U-Net discriminator. This approach processes 4D fMRI data to retain both spatial and temporal features while effectively mitigating noise. The 3D dense U-Net discriminator captures both global and local patterns, and the inclusion of adversarial and perceptual losses helps prevent oversmoothing and preserve structural integrity. This denoising step enhances downstream processing, including affine registration, by providing cleaner and more reliable fMRI data.</p>
<p>Motion correction is an essential preprocessing step in fMRI analysis to reduce the impact of subject movement during image acquisition. In our pipeline, we applied AFNI&#x00027;s rigid-body motion correction method (<xref ref-type="bibr" rid="B8">Cox, 1996</xref>) prior to skull stripping to ensure temporal alignment of the brain volumes and improve the accuracy of subsequent steps.</p>
<p>Prior to affine registration, skull stripping as illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref> is applied to remove non-brain tissues and improve anatomical alignment. Manual skull stripping is time-consuming and prone to variability, especially in preclinical fMRI data, which present challenges such as low resolution, varying slice sizes, and anatomical differences. To address these issues, we incorporate our recently developed SST-DUNet method (<xref ref-type="bibr" rid="B42">Soltanpour et al., 2025b</xref>), which combines a dense U-Net architecture with a Smart Swin Transformer-based feature extractor (<xref ref-type="bibr" rid="B14">Fu et al., 2024</xref>). The Smart Shifted Window Multi-Head Self-Attention (SSW-MSA) module enables robust feature learning by focusing on channel-wise dependencies within brain structures. Additionally, a hybrid loss function combining Focal and Dice loss mitigates class imbalance, resulting in more accurate skull extraction. This automated approach ensures reliable brain masking, which is essential for accurate affine registration.</p>
</sec>
</sec>
<sec>
<label>2.2</label>
<title>Proposed affine registration algorithm</title>
<p>The framework of the proposed affine registration method is presented in <xref ref-type="fig" rid="F2">Figure 2</xref>. Inspired by the Coarse-to-Fine Vision Transformer (C2FViT) proposed by <xref ref-type="bibr" rid="B32">Mok and Chung (2022)</xref> for affine registration of clinical MRI, our method follows a multi-stage hierarchical approach to solve affine registration using an image pyramid. Unlike standard Vision Transformers (ViT) (<xref ref-type="bibr" rid="B12">Dosovitskiy et al., 2020</xref>), which rely on self-attention over fixed patches and struggle with local feature extraction, we incorporate Swin Transformer (SwinT) blocks (<xref ref-type="bibr" rid="B29">Liu et al., 2021</xref>) alongside dilated convolutional block (<xref ref-type="bibr" rid="B16">Gao et al., 2022</xref>) to enhance feature representation and spatial awareness.</p>
<fig position="float" id="F2">
<label>Figure 2</label>
<caption><p>Overview of the MsDCSwinT affine registration algorithm. The model contains three stages to extract the final affine matrix.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1621244-g0002.tif">
<alt-text content-type="machine-generated">Diagram showcasing a multi-stage transformation process involving images F1, F2, and F3. Each stage involves fixed and moving images undergoing patch embedding, merging, and Swin Transformer operations. Upsampling and dilation occur, with annotated operations like position embedding and affine transformation at each stage.</alt-text>
</graphic>
</fig>
<p>While C2FViT utilizes convolutional patch embedding to encode local features at the input stage, it relies solely on transformer-based global attention throughout the network, lacking additional mechanisms for preserving local spatial relationships during deeper stages. In contrast, our model applies linear patch embedding followed by Swin Transformer blocks, which capture local features through windowed attention mechanisms. Importantly, we apply a dilated convolutional SwinT block after each stage, which explicitly enhances local feature modeling by expanding the receptive field without sacrificing spatial resolution. This architectural design allows our design to jointly model local anatomical variations and global structural alignments throughout all network depths, improving registration performance particularly in preclinical fMRI data. In contrast to ViT, SwinT employs a hierarchical structure to address dense prediction tasks while lowering computational costs. It achieves this by computing self-attention within non-overlapping windows of limited size. Additionally, to capture contextual details, the window configurations vary across successive layers. As a result, the network processes broader contextual information through localized self-attention mechanisms.</p>
<p>Our framework consists of three stages, each maintaining a similar architecture comprising a patch embedding layer, SwinT-based encoder and dilated convolutional blocks. In the first stage, a shallow encoder is employed to extract coarse structural information. In subsequent stages, the encoder depth increases to accommodate higher-resolution inputs, ensuring robust multi-scale feature learning. By leveraging the strengths of SwinT&#x00027;s hierarchical windowed attention alongside dilated convolutional blocks, our approach mitigates the limitations of ViT in capturing spatial dependencies and preserving the spatial integrity of functional regions, leading to more precise affine transformations for preclinical fMRI registration. Unlike the ViT, the SwinT employs a hierarchical structure that is well-suited for spatially dense tasks such as affine registration, while also reducing computational overhead. It performs self-attention within small, non-overlapping windows, allowing for efficient local feature extraction. To capture broader spatial context, the window partitions are shifted across successive layers, enabling the network to progressively model long-range dependencies through a series of local self-attention operations across the entire image volume.</p>
<p>In the proposed model, global spatial relationships are effectively captured through the shifted window self-attention mechanism of the Swin Transformer blocks. By progressively shifting the window partitions between layers, the model enables information flow across non-local regions, thereby modeling long-range dependencies across the full image. Additionally, the hierarchical multi-scale structure, which processes input volumes at varying resolutions, further enhances the network&#x00027;s ability to capture global spatial context by operating at progressively coarser scales where individual windows encompass larger anatomical regions.</p>
<p>Let <italic>F</italic> and <italic>M</italic> denote the fixed and moving 3D volumes, respectively, defined over a spatial domain &#x003A9;&#x02286;&#x0211D;<sup>3</sup>. This study aims to learn an optimal affine transformation matrix that aligns <italic>F</italic> and <italic>M</italic>. The affine registration is formulated as a learnable function <italic>f</italic><sub>&#x003B8;</sub>(<italic>F, M</italic>)=<italic>A</italic>, where &#x003B8; represents the set of trainable parameters and <italic>A</italic> is the resulting affine transformation matrix. To enable multi-stage learning, an input pyramid is constructed by downsampling <italic>F</italic> and <italic>M</italic> using trilinear interpolation, yielding scaled versions <italic>F</italic><sub><italic>i</italic></sub>&#x02208;{<italic>F</italic><sub>1</sub>, <italic>F</italic><sub>2</sub>, <italic>F</italic><sub>3</sub>} and <italic>M</italic><sub><italic>i</italic></sub>&#x02208;{<italic>M</italic><sub>1</sub>, <italic>M</italic><sub>2</sub>, <italic>M</italic><sub>3</sub>}. Each <italic>F</italic><sub><italic>i</italic></sub> and <italic>M</italic><sub><italic>i</italic></sub> corresponds to a downsampled version of <italic>F</italic> and <italic>M</italic> at a scale of 0.5<sup>(3&#x02212;<italic>i</italic>)</sup>.</p>
<sec>
<label>2.2.1</label>
<title>Patch embedding and merging blocks</title>
<p>For the fixed and moving images with the size of <italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>D</italic>, where <italic>H, W</italic>, and <italic>D</italic> are the image spatial dimension, we calculate the patch embeddings. In the first stage, where the input images have smaller resolution, we use a window of size 2 &#x000D7; 2 &#x000D7; 2. In the next stages, by increasing the resolutions, windows of size 4 &#x000D7; 4 &#x000D7; 4 and 8 &#x000D7; 8 &#x000D7; 8 are used to capture the larger patches. For the data with size <italic>H</italic>&#x000D7;<italic>W</italic>&#x000D7;<italic>D</italic>, we obtain vectors with the same number of features but with various feature lengths. In this way, for both <italic>F</italic><sub><italic>i</italic></sub> and <italic>M</italic><sub><italic>i</italic></sub>, we extract patches with length <italic>h</italic>&#x000D7;<italic>w</italic>&#x000D7;<italic>d</italic>&#x000D7;8, <italic>h</italic>&#x000D7;<italic>w</italic>&#x000D7;<italic>d</italic>&#x000D7;64, and <italic>h</italic>&#x000D7;<italic>w</italic>&#x000D7;<italic>d</italic>&#x000D7;512 for three different stages. We employ fully connected layers to convert the variable-length feature representations into fixed-size vectors of dimension <italic>C</italic>, ensuring uniformity in the output dimensions. Consequently, the general feature map is defined by <inline-formula><mml:math id="M1"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>w</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>d</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>, and <inline-formula><mml:math id="M2"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>w</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>d</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>C</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> for the fixes and moving images.</p>
<p>The patch merging operation, applied after each patch embedding and within each SwinT block, follows the same approach used in ViT. This mechanism is employed to create connections between non-overlapping image patches. Initially, the patch merging layer combines the features of adjacent 2 &#x000D7; 2 patches, resulting in a 4C-dimensional representation. This aggregated feature set is then passed through a linear layer, which not only reduces the number of tokens by a factor of four, equivalent to a twofold decrease in spatial resolution, but also transforms the feature representation accordingly. In this way, <inline-formula><mml:math id="M3"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> and <inline-formula><mml:math id="M4"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> are transformed to the same dimension as &#x0211D;<sup><italic>h</italic>/2 &#x000D7; <italic>w</italic>/2 &#x000D7; <italic>d</italic>/2 &#x000D7; 2<italic>C</italic></sup></p></sec>
<sec>
<label>2.2.2</label>
<title>Dilated convolutional Swin Transformer</title>
<p>To further enhance the affine registration performance, we integrate the dilated convolutional block, as proposed by <xref ref-type="bibr" rid="B16">Gao et al. (2022)</xref>, into our model. This design combines dilated convolutions with the SwinT architecture, enabling the model to capture both local and global contextual information across images. By leveraging dilated convolutions, we expand the receptive field without increasing computational complexity, which allows for more accurate localization of features in both the fixed and moving images. This is particularly advantageous for preclinical fMRI data, where image resolution and variability in structural features present significant challenges. Additionally, the SwinT&#x00027;s window-based self-attention mechanism facilitates capturing long-range dependencies, further improving the registration accuracy. Integrating the dilated convolutional block provides a robust approach to handling complex variations in brain geometry and resolution, ultimately enhancing the precision and reliability of the affine registration process.</p>
<p>In our affine Swin Transformer framework, we incorporate dilated convolution to effectively expand the receptive field across multiple stages without reducing spatial resolution. Dilated convolution proposed by <xref ref-type="bibr" rid="B47">Yu and Koltun (2015)</xref>, enables wider spatial context modeling compared to traditional convolutions. For example, while a standard 3 &#x000D7; 3 &#x000D7; 3 kernel captures local features within a limited 3 &#x000D7; 3 &#x000D7; 3 region, a 2-dilated convolution with the same kernel expands the receptive field to 7 &#x000D7; 7 &#x000D7; 7. This expansion is particularly advantageous in our model, where 3D inputs are processed through three hierarchical stages. By using dilated convolution at each stage, we enhance the model&#x00027;s ability to capture global affine transformations and structural dependencies across the full 3D volume, while maintaining spatial detail and resolution.</p>
<p>Considering that the data flow in SwinT uses vectors instead of feature maps, as in traditional CNNs, the dilated convolution block first reshapes a group of vector features into a spatial feature map. For instance, a set of tokens with dimensions <inline-formula><mml:math id="M5"><mml:mfrac><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula> is reshaped into a feature map of size <inline-formula><mml:math id="M6"><mml:mfrac><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mi>C</mml:mi></mml:math></inline-formula> where <italic>p</italic> is the patch size. Following this, two dilated convolutional layers with Batch Normalization (<xref ref-type="bibr" rid="B24">Ioffe and Szegedy, 2015</xref>) and ReLU activation are applied to capture large-range spatial features.Finally, the feature map is re-transformed back into the original token dimensions <inline-formula><mml:math id="M7"><mml:mfrac><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x000D7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula> and passed to the next stage in the model. This integration of dilated convolutions with SwinT allows the model to capture both local and global spatial features, improving the overall accuracy and robustness of affine registration, particularly in challenging preclinical fMRI datasets.</p></sec>
<sec>
<label>2.2.3</label>
<title>Affine transformation</title>
<p>The hierarchical self attention mechanism of the SwinT (<xref ref-type="bibr" rid="B29">Liu et al., 2021</xref>) is highly effective to model long range dependencies within sequences of embeddings. In our algorithm, the misalignment between fixed and moving images is captured through two consecutive SwinT blocks including W-MSA and SW-MSA which are multi-head self attention modules with regular and shifted windowing configurations, respectively. The structure of SwinT block has been shown in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig position="float" id="F3">
<label>Figure 3</label>
<caption><p>Two consecutive Swin Transformer blocks are used, incorporating W-MSA and SW-MSA, which are multi-head self-attention mechanisms employing regular and shifted window configurations, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1621244-g0003.tif">
<alt-text content-type="machine-generated">Diagram showing a transformer block structure with two sections. The first section includes a Multi-Layer Perceptron (MLP), Layer Normalization, Window-based Multi-Head Self-Attention (W-MSA), and another Layer Normalization. The second section mirrors the first but uses Shifted Window-based Multi-Head Self-Attention (SW-MSA). Both sections include residual connections denoted by &#x0201C;+&#x0201D;.</alt-text>
</graphic>
</fig>
<p>The formula to represent the SwinT block is as follows:</p>
<disp-formula id="EQ1"><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>&#x01E90;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">MSA</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">LN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x01E90;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x01E90;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">MLP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">LN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x01E90;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x01E90;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>&#x01E90;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">MSA</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">LN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>Z</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">MLP</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mtext class="textrm" mathvariant="normal">LN</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x01E90;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x01E90;</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(1)</label></disp-formula>
<p>where &#x01E90;<sup><italic>l</italic></sup> and <italic>Z</italic><sup><italic>l</italic></sup> denote the output of block <italic>l</italic>. <italic>LN</italic> (LayerNorm) represents a regularization. <italic>MLP</italic> denotes multi-layer preceptron. <italic>W</italic><sub><italic>MSA</italic></sub> and <italic>SW</italic><sub><italic>MSA</italic></sub> represent the window self-attentive mechanism and the shifted window self-attentive respectively. Finally, the attention head&#x00027;s output is input to the dilated convultional block and the output is passed through a multi-layer linear network, which generates the corresponding affine transformation matrix <italic>A</italic><sub><italic>i</italic></sub>.</p>
<p>The transformer encoders in MsDCSwinT use the similarity between projected query-key pairs to capture misalignment and global relationships between the fixed and moving images, generating attention scores for each patch embedding. The query (<italic>Q</italic>), key (<italic>K</italic>), and value (<italic>V</italic>) are linearly projected from the patch embeddings (tokens).</p>
<disp-formula id="EQ2"><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>Q</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x01E90;</mml:mi><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>Q</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x01E90;</mml:mi><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mi>V</mml:mi><mml:mo>=</mml:mo><mml:mi>&#x01E90;</mml:mi><mml:msup><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(2)</label></disp-formula>
<p>Considering that the number of attention heads in the SwinT block is <italic>h</italic>&#x02032;, the linear projection matrices are <inline-formula><mml:math id="M10"><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>Q</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>W</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>V</mml:mi></mml:mrow></mml:msup><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>D</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula>, and <inline-formula><mml:math id="M11"><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>D</mml:mi><mml:mo>/</mml:mo><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. Attention operation for attention head <italic>j</italic> is calculated as follows:</p>
<disp-formula id="EQ3"><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Attention</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">Softmax</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>V</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(3)</label></disp-formula>
<p>where <inline-formula><mml:math id="M13"><mml:msub><mml:mrow><mml:mi>D</mml:mi></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x02032;</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:math></inline-formula> is the embedding dimension and <italic>B</italic><sub><italic>P</italic></sub> denotes the relative position encoding. In this study, we employ <italic>h</italic>&#x02032;=2 attention heads for all the transformer encoders. Finally, the attended embeddings from all attention heads are concatenated and passed through a linear projection matrix.</p></sec>
<sec>
<label>2.2.4</label>
<title>Multi-stage affine transformation estimation</title>
<p>We incorporate a multiresolution approach into our architecture. Specifically, each stage of MsDCSwinT includes a classification head, which consists of two consecutive MLP layers with a hyperbolic tangent (Tanh) activation function. This classification head processes the averaged patch-wise embeddings and generates a set of affine parameters. At each intermediate stage <italic>i</italic>, the resulting affine matrix is applied to progressively transform the moving image <italic>M</italic><sub><italic>i</italic>&#x0002B;1</sub> via a warping operation using a spatial transformer (<xref ref-type="bibr" rid="B25">Jaderberg et al., 2015</xref>). The warped image <italic>M</italic><sub><italic>i</italic>&#x0002B;1</sub> is then concatenated with the fixed image <italic>F</italic><sub><italic>i</italic>&#x0002B;1</sub> and passed to the next stage, <italic>i</italic>&#x0002B;1. This progressive transformation strategy allows for initial misalignments to be corrected at lower resolutions, enabling higher-level transformers to focus on more complex misalignments, thus simplifying the problem at later stages.</p></sec>
<sec>
<label>2.2.5</label>
<title>Geometric transformation model</title>
<p>Rather than directly estimating the affine transformation matrix, our model predicts a set of geometric transformation parameters. Specifically, the affine registration problem is reformulated as:</p>
<disp-formula id="EQ4"><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>t</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>r</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>s</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>h</mml:mtext></mml:mstyle></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(4)</label></disp-formula>
<p>We parameterize the affine transform with [<italic>t, r, s, h</italic>]&#x02208;&#x0211D;<sup>12</sup>, where <italic>t</italic>=(<italic>t</italic><sub><italic>x</italic></sub>, <italic>t</italic><sub><italic>y</italic></sub>, <italic>t</italic><sub><italic>z</italic></sub>), <italic>r</italic>=(<italic>r</italic><sub><italic>x</italic></sub>, <italic>r</italic><sub><italic>y</italic></sub>, <italic>r</italic><sub><italic>z</italic></sub>), <italic>s</italic>=(<italic>s</italic><sub><italic>x</italic></sub>, <italic>s</italic><sub><italic>y</italic></sub>, <italic>s</italic><sub><italic>z</italic></sub>), and <italic>h</italic>=(<italic>h</italic><sub><italic>xy</italic></sub>, <italic>h</italic><sub><italic>xz</italic></sub>, <italic>h</italic><sub><italic>yz</italic></sub>). Using homogeneous coordinates, the overall affine matrix <bold>A</bold> is the ordered product:</p>
<disp-formula id="EQ5"><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>A</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>T</mml:mtext></mml:mstyle><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>R</mml:mtext></mml:mstyle><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>S</mml:mtext></mml:mstyle><mml:mo>&#x000B7;</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>H</mml:mtext></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(5)</label></disp-formula>
<p>where <italic>T</italic>, <italic>R</italic>, <italic>S</italic>, and <italic>H</italic> represent translation, rotation, scaling, and shearing matrices, respectively.</p>
<p>Translation (<italic>T</italic>):</p>
<disp-formula id="EQ6"><mml:math id="M16"><mml:mrow><mml:mi>T</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Rotation (<italic>R</italic>): The overall rotation matrix is defined as:</p>
<disp-formula id="EQ7"><mml:math id="M17"><mml:mrow><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Rotation about <italic>x</italic>-axis:</p>
<disp-formula id="EQ8"><mml:math id="M18"><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mo class="qopname">cos</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>-</mml:mo><mml:mo class="qopname">sin</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mo class="qopname">sin</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo class="qopname">cos</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Rotation about <italic>y</italic>-axis:</p>
<disp-formula id="EQ9"><mml:math id="M19"><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mo class="qopname">cos</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mo class="qopname">sin</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo>-</mml:mo><mml:mo class="qopname">sin</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mo class="qopname">cos</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Rotation about <italic>z</italic>-axis:</p>
<disp-formula id="EQ10"><mml:math id="M20"><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mo class="qopname">cos</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo>-</mml:mo><mml:mo class="qopname">sin</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo class="qopname">sin</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mo class="qopname">cos</mml:mo><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Scaling (<italic>S</italic>):</p>
<disp-formula id="EQ11"><mml:math id="M21"><mml:mrow><mml:mi>S</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>Shearing (<italic>H</italic>):</p>
<disp-formula id="EQ12"><mml:math id="M22"><mml:mrow><mml:mi>H</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mtable style="text-align:axis;" equalrows="false" columnlines="none none none none none none none none none" equalcolumns="false" class="array"><mml:mtr><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:msub><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi><mml:mi>z</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>0</mml:mn></mml:mtd><mml:mtd><mml:mn>1</mml:mn></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula>
<p>To reduce the search space and enforce meaningful transformations, we constrain the predicted parameters during training as follows: rotation and shearing values are limited to the range [&#x02212;&#x003C0;, &#x003C0;], translation values are constrained within &#x000B1;50% of the image resolution, and scaling values are restricted to the range [0.5, 1.5]. Additionally, we apply the center of mass of the image <bold>c</bold><sub><italic>I</italic></sub> (<xref ref-type="bibr" rid="B32">Mok and Chung, 2022</xref>) computed as:</p>
<disp-formula id="EQ13"><mml:math id="M23"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>c</mml:mtext></mml:mstyle></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mtext>&#x003A9;</mml:mtext></mml:mrow></mml:msub></mml:mstyle><mml:mi>p</mml:mi><mml:mo>&#x000B7;</mml:mo><mml:mi>I</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mtext>&#x003A9;</mml:mtext></mml:mrow></mml:msub></mml:mstyle><mml:mi>I</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(6)</label></disp-formula>
<p>where &#x003A9; represents the spatial domain of the image and <italic>p</italic> denotes the voxel position.</p></sec>
<sec>
<label>2.2.6</label>
<title>Loss function</title>
<p>Our affine registration method is based on unsupervised learning. This is primarily because generating manual segmentation maps for fMRI data is extremely time-consuming and impractical. Additionally, creating manual maps using the mean of fMRI data often results in inconsistent segmentation, particularly due to the low resolution of the data, which can lead to varying numbers of detected regions across different samples. Given these challenges and the unavailability of reliable ground truth labels, we opted for an unsupervised approach that does not rely on manually annotated data. Instead, the model learns to align the input image with an atlas by optimizing a similarity measure between them, allowing for robust and automated registration even in the absence of labelled training data.</p>
<p>The affine registration problem is parametrized as a learning problem to minimize the following equation:</p>
<disp-formula id="EQ14"><mml:math id="M24"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msup><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow><mml:mrow><mml:mo>*</mml:mo></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x003B8;</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x1D53C;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:mi>D</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(7)</label></disp-formula>
<p>where &#x003B8; represents the learning parameters in the MsDCSwinT model, <italic>F</italic> is the atlas, and <italic>M</italic> is the moving image from the training dataset <italic>D</italic>. The loss function <inline-formula><mml:math id="M25"><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:math></inline-formula> measures the similarity between the atlas <italic>F</italic> and the affine transformed moving image <italic>M</italic>(&#x003D5;(<italic>A</italic>)). We use the negative Normalized Cross-Correlation (NCC) (<xref ref-type="bibr" rid="B33">Mok and Chung, 2020b</xref>) similarity measure to quantify the distance between <italic>F</italic> and <italic>M</italic>(&#x003D5;(<italic>A</italic>)) as follows:</p>
<disp-formula id="EQ15"><mml:math id="M26"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">sim</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mi>L</mml:mi></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow></mml:munder></mml:mstyle><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>L</mml:mi><mml:mo>-</mml:mo><mml:mi>i</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">NCC</mml:mtext></mml:mrow><mml:mrow><mml:mi>w</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>M</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(8)</label></disp-formula>
<p>where <italic>L</italic> represents the number of image pyramid levels, <italic>NCC</italic><sub><italic>w</italic></sub> denotes the local normalized cross-correlation with window size <italic>w</italic>&#x000D7;<italic>w</italic>, and (<italic>F</italic><sub><italic>i</italic></sub>, <italic>M</italic><sub><italic>i</italic></sub>) are the atlas and moving images in the image pyramid.</p>
<p>The overall loss function is inspired by the energy-based formulation used in traditional image registration. It consists of two components: a similarity loss between the fixed image (atlas) and the affine-transformed moving image, and a regularization term that constrains the affine parameters to prevent implausible transformations. The total loss is defined as:</p>
<disp-formula id="EQ16"><mml:math id="M27"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">sim</mml:mtext></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mo>,</mml:mo><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003BB;</mml:mi><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(9)</label></disp-formula>
<p>where <italic>F</italic> is the fixed image (atlas), <italic>M</italic> is the moving image, and &#x003D5; denotes the affine transformation predicted by the network. The similarity term <inline-formula><mml:math id="M28"><mml:msub><mml:mrow><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>L</mml:mi></mml:mstyle></mml:mrow></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">sim</mml:mtext></mml:mstyle></mml:mrow></mml:msub></mml:math></inline-formula> quantifies alignment quality, for which we use a multi-resolution negative normalized cross-correlation (NCC) as described in <xref ref-type="disp-formula" rid="EQ15">Equation 8</xref>.</p>
<p>The regularization term <inline-formula><mml:math id="M29"><mml:mrow><mml:mstyle mathvariant="script"><mml:mi>R</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> penalizes overly large or unrealistic affine transformations to maintain stable optimization. In our work, we regularize the rotation and shearing parameters by constraining them within [&#x02212;&#x003C0;, &#x0002B;&#x003C0;], translation within [&#x02212;0.5<italic>R</italic>, &#x0002B;0.5<italic>R</italic>], and scaling between [0.5, 1.5], where <italic>R</italic> denotes the spatial resolution of the input image. This encourages the model to learn physically plausible transformations while maintaining registration accuracy. Here, &#x003BB; is a regularization weighting parameter that balances the contribution of the similarity loss and the regularization term. A higher value of &#x003BB; enforces stronger constraints on the transformation parameters, while a lower value focuses more on image similarity. In our experiments, we empirically set &#x003BB;=0.01 to achieve a good balance between accuracy and stability.</p></sec>
<sec>
<label>2.2.7</label>
<title>Implementation details</title>
<p>We applied the mean of fMRI scans to create 3D MRI data. We resampled and padded all scans to 96 &#x000D7; 96 &#x000D7; 48 with the same resolution (0.3mm &#x000D7; 0.3mm &#x000D7; 0.4mm). The input voxel values were adjusted to fall within the range of 0.0&#x02013;1.0 through normalization. The model was implemented using the PyTorch framework and trained on an Nvidia GeForce RTX 4090 GPU with CUDA support. We used the Adam optimizer with a fixed learning rate of 0.0001 and a batch size of 1. The value of &#x003BB;=0.01 was selected based on validation performance to ensure optimal registration accuracy without overfitting the transformation parameters.</p>
</sec>
</sec>
<sec>
<label>2.3</label>
<title>Functional connectivity analysis</title>
<sec>
<label>2.3.1</label>
<title>Preprocessing</title>
<p>All fMRI datasets were preprocessed using AFNI (<ext-link ext-link-type="uri" xlink:href="https://afni.nimh.nih.gov/">https://afni.nimh.nih.gov/</ext-link>). Each image volume was smoothed using a Gaussian kernel with a full width at half maximum (FWHM) of 0.1 mm. The data were then cropped using a previously generated brain mask, aligned to a down-sampled rat 96 &#x000D7; 96 &#x000D7; 48 template space, and band-pass filtered within the frequency range of 0.01&#x02013;0.1 Hz to remove physiological and low-frequency noise artifacts. These steps follow established preprocessing protocols (<xref ref-type="bibr" rid="B30">Lupinsky et al., 2025</xref>; <xref ref-type="bibr" rid="B44">Sourty et al., 2024b</xref>,<xref ref-type="bibr" rid="B43">a</xref>; <xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>; <xref ref-type="bibr" rid="B28">Karatas et al., 2021</xref>).</p></sec>
<sec>
<label>2.3.2</label>
<title>Post-processing and functional connectivity analysis</title>
<p>Post-processing included seed-based functional connectivity (FC) analysis, one of the most widely adopted methods in rodent fMRI studies (<xref ref-type="bibr" rid="B30">Lupinsky et al., 2025</xref>; <xref ref-type="bibr" rid="B43">Sourty et al., 2024a</xref>,<xref ref-type="bibr" rid="B44">b</xref>; <xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>; <xref ref-type="bibr" rid="B28">Karatas et al., 2021</xref>; <xref ref-type="bibr" rid="B38">Nasseef et al., 2019</xref>). Seed-to-seed correlation analysis was conducted on datasets from eight rats, all of which underwent identical preprocessing pipelines, except for the registration methods. Based on these methods, five registration groups were established including MsDCSwinT (ours), ANTS, C2FViT, C2FGALF, and ConvNet.</p></sec>
<sec>
<label>2.3.3</label>
<title>Atlas-based region definition</title>
<p>To define anatomical brain regions, we utilized a previously generated, in-house high-resolution rat brain atlas (<xref ref-type="bibr" rid="B46">Wang et al., 2025</xref>; <xref ref-type="bibr" rid="B15">Fuini et al., 2025</xref>), initially composed of 176 brain regions at a spatial resolution of 256 &#x000D7; 256 &#x000D7; 64. To ensure compatibility between the lower-resolution functional imaging data and the high-resolution rat brain atlas, the atlas was down-sampled to a spatial resolution of 96 &#x000D7; 96 &#x000D7; 48, yielding a total of 159 functional brain regions.</p></sec>
<sec>
<label>2.3.4</label>
<title>Seed mask construction and connectivity computation</title>
<p>Seed regions were defined based on the down-sampled atlas. The mean time series for each seed region was extracted, and pairwise Pearson correlations were computed and Fisher&#x00027;s Z transformation was applied to generate the seed-to-seed FC matrix. Statistical significance of connectivity differences between groups was assessed using two-sample t-tests with False Discovery Rate (FDR) correction (<xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>, <xref ref-type="bibr" rid="B38">2019</xref>, <xref ref-type="bibr" rid="B36">2018</xref>). Statistical significance of connectivity differences between groups was assessed using two-sample t-tests with False Discovery Rate (FDR) correction (<xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>, <xref ref-type="bibr" rid="B36">2018</xref>, <xref ref-type="bibr" rid="B38">2019</xref>)). MATLAB-based Multiple Testing Toolbox (<ext-link ext-link-type="uri" xlink:href="https://www.mathworks.com/matlabcentral/fileexchange/70604-multiple-testing-toolbox">https://www.mathworks.com/matlabcentral/fileexchange/70604-multiple-testing-toolbox</ext-link>) was used for circular plotting and FDR correction. All computations were carried out using an in-house developed MATLAB script, following our previously validated pipeline (<xref ref-type="bibr" rid="B30">Lupinsky et al., 2025</xref>; <xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>, <xref ref-type="bibr" rid="B38">2019</xref>).</p></sec></sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Experimental results</title>
<sec>
<label>3.1</label>
<title>Affine registration performance analysis</title>
<p>To evaluate the performance of our proposed MsDCSwinT model in preclinical fMRI affine registration, a comparative analysis was conducted against existing methods. The evaluated methods included an iterative technique implemented in ANTS (<xref ref-type="bibr" rid="B2">Avants et al., 2011</xref>), and learning-based affine methods ConvNet (<xref ref-type="bibr" rid="B9">De Vos et al., 2019</xref>), C2FViT (<xref ref-type="bibr" rid="B32">Mok and Chung, 2022</xref>), and C2FGALF (<xref ref-type="bibr" rid="B27">Ji and Yang, 2024</xref>) developed for clinical image registration.</p>
<p>We use the affine registration implementation provided in the publicly available ANTS software package, which adopts a three-level multi-resolution optimization framework based on adaptive gradient descent and mutual information as the similarity metric. For learning-based methods, we follow the parameter settings recommended in their respective publications. All models are trained in an unsupervised manner using the similarity defined in <xref ref-type="disp-formula" rid="EQ15">Equation 8</xref>.</p>
<p>To enable robust affine registration of 4D fMRI data, we first reduced the dimensionality of the training inputs by computing the mean across the time dimension. Specifically, for each 4D fMRI dataset, we averaged the voxel intensities over all time points to generate a representative 3D MRI volume. This 3D mean image captures the essential anatomical structure while reducing the impact of temporal fluctuations, facilitating more stable training of the registration network. During inference, the trained network was applied to the 3D mean image of each subject in the CTNI fMRI test set to estimate the corresponding affine transformation matrix. This affine matrix was applied uniformly to register all individual 3D volumes across the 295 time points of the 4D fMRI data. This approach ensures consistent spatial alignment throughout the entire fMRI time series.</p>
<sec>
<label>3.1.1</label>
<title>Evaluation metrics</title>
<p>To evaluate the performance of our affine registration algorithm, we align each subject&#x00027;s image to the atlas. The registration accuracy is assessed using the Dice Similarity Coefficient (DSC) (<xref ref-type="bibr" rid="B11">Dice, 1945</xref>), which quantifies the overlap between the transformed moving image and the atlas. This provides a measure of segmentation accuracy. In addition to DSC, we compute the 95% percentile of the Hausdorff Distance (HD95) (<xref ref-type="bibr" rid="B21">Huttenlocher et al., 1993</xref>), which measures the distance between the boundaries of the transformed moving image and the atlas. Together, these metrics offer a comprehensive view of the registration&#x00027;s precision and reliability.</p>
<p>The Dice Similarity Coefficient (DSC) for the subcortical segmentation map can be formulated as:</p>
<disp-formula id="EQ17"><mml:math id="M30"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Dice</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x02229;</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo><mml:mo>&#x0002B;</mml:mo><mml:mo>|</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>|</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(10)</label></disp-formula>
<p>where <inline-formula><mml:math id="M31"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> represents the set of voxels of structure <italic>k</italic> in the registered moving image, and <inline-formula><mml:math id="M32"><mml:mrow><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:math></inline-formula> represents the set of voxels of structure <italic>k</italic> in the fixed image (atlas).</p>
<p>Additionally, we evaluate the 95th percentile of the Hausdorff distance (HD95) between segmentation maps to assess the robustness of the registration algorithm. The calculation can be formulated as:</p>
<disp-formula id="EQ18"><mml:math id="M33"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext class="textrm" mathvariant="normal">HD</mml:mtext></mml:mrow><mml:mrow><mml:mn>95</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mo class="qopname">95%</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mtext>&#x000A0;</mml:mtext><mml:mo>||</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math><label>(11)</label></disp-formula>
<p>where</p>
<disp-formula id="EQ19"><mml:math id="M34"><mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">min</mml:mo></mml:mrow><mml:mrow><mml:mi>b</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder></mml:mstyle><mml:mo>||</mml:mo><mml:mi>a</mml:mi><mml:mo>-</mml:mo><mml:mi>b</mml:mi><mml:msub><mml:mrow><mml:mo>||</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:mi>a</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<disp-formula id="EQ20"><mml:math id="M35"><mml:mrow><mml:mi>d</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">min</mml:mo></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003D5;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow></mml:munder></mml:mstyle><mml:mo>||</mml:mo><mml:mi>b</mml:mi><mml:mo>-</mml:mo><mml:mi>a</mml:mi><mml:msub><mml:mrow><mml:mo>||</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>&#x02223;</mml:mo><mml:mi>b</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msubsup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula>
<p>and ||&#x000B7;||<sub>2</sub> denotes the Euclidean distance.</p>
<p>In addition to reporting the traditional overall Dice Similarity Coefficient (DSC) to evaluate our model&#x00027;s performance, we further assessed the registration accuracy by analyzing the distribution of results. Specifically, we computed the 30% lowest DSC of all cases (DSC30), based on the registration accuracy ranking across the 159 segmented structures in the test set. To provide a more complete analysis, we also reported the DSC for unregistered data (moving data). Together, these evaluations offer a comprehensive understanding of the registration accuracy and robustness of the proposed method.</p></sec>
<sec>
<label>3.1.2</label>
<title>Quantitative and qualitative affine registration evaluation results</title>
<p>The overall experimental results are summarized in <xref ref-type="table" rid="T1">Table 1</xref>. Our proposed method, MsDCSwinT, achieves a DSC of 0.9286 and 0.9214 on the CTNI dataset1 and dataset 2, demonstrating the best registration performance among all methods compared. The most competing conventional method, ANTS, achieves a DSC of 0.9159 and 0.9082 on dataset 1 and dataset 2 respectively. Other learning-based methods, including ConvNet, C2FViT, and C2FGALF achieve DSCs of 0.7469, 0.8408, and 0.8638 for dataset 1, and 0.7294, 0.8231, and 0.8469 for dataset 2 respectively. In comparison, our model, MsDCSwinT, not only achieves an overall registration accuracy of 0.93 but also obtains the highest DSC30 and lowest HD95 outperforming all competing methods on both datasets.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Quantitative evaluation of the results on the CTNI dataset.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>DSC</bold></th>
<th valign="top" align="center"><bold>DSC30</bold></th>
<th valign="top" align="center"><bold>HD95</bold></th>
</tr>
<tr>
<th valign="top" align="left" colspan="4"><bold>Dataset1</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Unregistered</td>
<td valign="top" align="center">0.4883 &#x000B1; 0.1274<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.3317 &#x000B1; 0.0437<sup>&#x0002A;</sup></td>
<td valign="top" align="center">18.9103 &#x000B1; 1.2315<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td valign="top" align="left">ANTS</td>
<td valign="top" align="center">0.9159 &#x000B1; 0.0329</td>
<td valign="top" align="center">0.8949 &#x000B1; 0.0233<sup>&#x0002A;</sup></td>
<td valign="top" align="center">2.0604 &#x000B1; 0.3152</td>
</tr>
<tr>
<td valign="top" align="left">ConvNet</td>
<td valign="top" align="center">0.7469 &#x000B1; 0.0687<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.6066 &#x000B1; 0.0501<sup>&#x0002A;</sup></td>
<td valign="top" align="center">8.4757 &#x000B1; 0.5335<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td valign="top" align="left">C2FViT</td>
<td valign="top" align="center">0.8408 &#x000B1; 0.0514<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.7315 &#x000B1; 0.0459</td>
<td valign="top" align="center">5.1041 &#x000B1; 0.5818<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td valign="top" align="left">C2FGALF</td>
<td valign="top" align="center">0.8638 &#x000B1; 0.0470<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.7627 &#x000B1; 0.0402</td>
<td valign="top" align="center">4.8079 &#x000B1; 0.4973<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td valign="top" align="left">MsDCSwinT (ours)</td>
<td valign="top" align="center"><bold>0.9286</bold> <bold>&#x000B1;0.0347</bold></td>
<td valign="top" align="center"><bold>0.9198</bold> <bold>&#x000B1;0.0287</bold></td>
<td valign="top" align="center"><bold>1.9535</bold> <bold>&#x000B1;0.3973</bold></td>
</tr>
<tr>
<td valign="top" align="left" colspan="4"><bold>Dataset2</bold></td>
</tr>
<tr>
<td valign="top" align="left">Unregistered</td>
<td valign="top" align="center">0.4625 &#x000B1; 0.1291<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.3252 &#x000B1; 0.0554<sup>&#x0002A;</sup></td>
<td valign="top" align="center">19.3821 &#x000B1; 1.3872<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td valign="top" align="left">ANTS</td>
<td valign="top" align="center">0.9082 &#x000B1; 0.0334</td>
<td valign="top" align="center">0.8983 &#x000B1; 0.0261<sup>&#x0002A;</sup></td>
<td valign="top" align="center">2.7472 &#x000B1; 0.3227</td>
</tr>
<tr>
<td valign="top" align="left">ConvNet</td>
<td valign="top" align="center">0.7294 &#x000B1; 0.0702<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.5901 &#x000B1; 0.0613<sup>&#x0002A;</sup></td>
<td valign="top" align="center">8.7963 &#x000B1; 0.5741<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td valign="top" align="left">C2FViT</td>
<td valign="top" align="center">0.8231 &#x000B1; 0.0527<sup>&#x0002A;</sup></td>
<td valign="top" align="center">0.7184 &#x000B1; 0.0547<sup>&#x0002A;</sup></td>
<td valign="top" align="center">5.3264 &#x000B1; 0.6228<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td valign="top" align="left">C2FGALF</td>
<td valign="top" align="center">0.8496 &#x000B1; 0.0565</td>
<td valign="top" align="center">0.7519 &#x000B1; 0.0498</td>
<td valign="top" align="center">5.0437 &#x000B1; 0.5081<sup>&#x0002A;</sup></td>
</tr>
<tr>
<td valign="top" align="left">MsDCSwinT (ours)</td>
<td valign="top" align="center"><bold>0.9214</bold> <bold>&#x000B1;0.0551</bold></td>
<td valign="top" align="center"><bold>0.9027</bold> <bold>&#x000B1;0.0292</bold></td>
<td valign="top" align="center"><bold>2.1129</bold> <bold>&#x000B1;0.2036</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Metric values are reported for the registered mean of fMRI data as mean &#x000B1; standard deviation.</p>
<p>The bolded numbers indicate the highest scores for each metric. Statistical analysis compares the performance of the proposed MsDCSwinT with other competing methods using a two-sided <italic>t</italic>-test, where <sup>&#x0002A;</sup><italic>p</italic>-value &#x0003C; 0.05 indicates statistical significance.</p>
</table-wrap-foot>
</table-wrap>
<p><xref ref-type="fig" rid="F4">Figure 4</xref> presents a qualitative comparison for the test set of fMRI dataset 1, highlighting the visual differences between the moving (unregistered) image, the registered images using different methods, and the atlas across representative slices and time points. As shown in <xref ref-type="fig" rid="F4">Figure 4</xref>, example coronal, axial, and sagittal slices from the test dataset 1 illustrate the visual alignment performance of various registration methods, including ANTS, ConvNet, C2FViT, C2FGALF, and our proposed MsDCSwinT model. The figure displays the fixed image (Atlas), unregistered images, and the corresponding warped outputs produced by each method. Color-coded overlays help visualize registration quality, green indicates atlas-only regions, red shows registered-only regions, and yellow highlights overlapping areas, which represent perfect alignment. Visually, our proposed MsDCSwinT achieves the highest degree of overlap with the atlas, indicated by more extensive yellow regions. While the performance of MsDCSwinT is very close to ANTS, a widely regarded traditional method, it clearly outperforms other deep learning-based approaches. These results demonstrate that our model effectively combines the robustness of traditional methods with the efficiency and automation of deep learning, achieving accurate registration while maintaining anatomical integrity. The distribution of evaluation metrics for CTNI fMRI test sets is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. All results have been computed as the mean &#x000B1; standard deviation over all slices and time points of the fMRI test sets.</p>
<fig position="float" id="F4">
<label>Figure 4</label>
<caption><p>Example coronal, axial, and sagittal fMR slices from the test dataset 1 are shown. The slices are taken from the fixed image (Atlas), moving images (Unregistered), and the resulting warped images produced by ANTS, ConvNet, C2FViT, C2FGALF, and our proposed MsDCSwinT. Atlas-only regions are shown in green, registered-only regions in red, and overlapping regions in yellow, indicating alignment between the atlas and registered image.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1621244-g0004.tif">
<alt-text content-type="machine-generated">Grid of brain scan images showing comparisons across different techniques including Atlas (Fixed), Unregistered (Moving), ANTS, ConvNet, C2FViT, C2FGALF, and MsDCSwinT. Rows depict colored overlays and grayscale scans for various slices.</alt-text>
</graphic>
</fig>
<fig position="float" id="F5">
<label>Figure 5</label>
<caption><p>Methods comparison on CTNI fMRI test dataset 1 and dataset 2. MsDCSwinT approach is superior than the alternative approaches in terms of DSC, DSC30, and HD95.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1621244-g0005.tif">
<alt-text content-type="machine-generated">Three bar charts compare different methods across two datasets. The top chart shows DSC scores where &#x0201C;MsDCSwinT (ours)&#x0201D; scores highest for both datasets. The middle chart presents DSC90, again with &#x0201C;MsDCSwinT (ours)&#x0201D; having the highest scores. The bottom chart shows HD95 values, where &#x0201C;Unregistered&#x0201D; has the highest values, and &#x0201C;MsDCSwinT (ours)&#x0201D; the lowest. Error bars indicate variability.</alt-text>
</graphic>
</fig>
<p><xref ref-type="table" rid="T2">Table 2</xref> presents the average inference times for all methods. We report the average registration time to register the mean of fMRI data to the atlas. As the table shows our method is the fastest among the methods evaluated, mainly due to GPU acceleration and efficient learning-based design. ConvNet, C2FViT, and C2FGALF are also significantly faster than ANTS. ANTS shows variable runtimes depending on the level of initial misalignment.Our model reduces the inference time to just 0.12 s, making it much more suitable for the practical demands of preclinical image registration.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Performance comparison.</p></caption>
<table frame="box" rules="all">
<thead>
<tr>
<th valign="top" align="left"><bold>Methods</bold></th>
<th valign="top" align="center"><bold>Parameters (M)</bold></th>
<th valign="top" align="center"><bold>Inference time (s)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ANTS</td>
<td valign="top" align="center">-</td>
<td valign="top" align="center">33.38 &#x000B1; 2.24</td>
</tr>
<tr>
<td valign="top" align="left">ConvNet</td>
<td valign="top" align="center">14.8</td>
<td valign="top" align="center">0.21 &#x000B1; 0.05</td>
</tr>
<tr>
<td valign="top" align="left">C2FViT</td>
<td valign="top" align="center">16.4</td>
<td valign="top" align="center">0.27 &#x000B1; 0.08</td>
</tr>
<tr>
<td valign="top" align="left">C2FGALF</td>
<td valign="top" align="center">15.7</td>
<td valign="top" align="center">0.16 &#x000B1; 0.1</td>
</tr>
<tr>
<td valign="top" align="left">MsDCSwinT (Ours)</td>
<td valign="top" align="center">15.2</td>
<td valign="top" align="center"><bold>0.12</bold> <bold>&#x000B1;0.03</bold></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The inference time is measured as the average runtime across results in the test set of dataset 1. The bold value indicates the best result.</p>
</table-wrap-foot>
</table-wrap>
<p>To provide a complete view of the computational efficiency, we report the total preprocessing time for each subject, which includes denoising, skull stripping, and affine registration. The full proposed pipeline requires approximately 24.85 s per subject, consisting of 24.18 s for denoising, 0.55 s for skull stripping, and about 0.12 s for affine registration. For comparison, ANTs registration alone takes about 33.38 s, which is already longer than the entire runtime of our full preprocessing workflow. These results show that our proposed registration method contributes less than 1% of the total processing time and that the overall pipeline remains substantially faster than ANTs-based preprocessing approaches.</p>
</sec>
</sec>
<sec>
<label>3.2</label>
<title>Functional connectivity analysis results</title>
<sec>
<label>3.2.1</label>
<title>Data visualization</title>
<p>To facilitate group-level interpretation, the results were visualized using MATLAB&#x00027;s built-in plotting tools. Visualization formats included scatter plots, and box plots. For circular plotting, the CircularGraph (<ext-link ext-link-type="uri" xlink:href="https://www.mathworks.com/matlabcentral/fileexchange/48576-circulargraph">https://www.mathworks.com/matlabcentral/fileexchange/48576-circulargraph</ext-link>) toolkit was applied. These representations provided intuitive insights into group-wise variability, inter-regional correlation patterns, and overall data distribution, thereby enhancing the interpretability of the findings.</p></sec>
<sec>
<label>3.2.2</label>
<title>Seed-to-Seed rs-fMRI reveals high similarity between proposed MsDCSwinT and other registration methods</title>
<p>Resting-state fMRI (rs-fMRI) data from all eight rats, preprocessed through five different automated registration methods (including the proposed MsDCSwinT), were analyzed across 159 functional brain regions. For each dataset, mean time series were extracted from each region based on the corresponding registration. This hypothesis-driven, atlas-based approach allowed for a comprehensive whole-brain investigation of connectivity differences between the MsDCSwinT method and the five other registration strategies. To evaluate inter-group connectivity patterns, symmetric Pearson&#x00027;s correlation coefficients (CC) were first computed for all pairs of the 159 regions within each rat and automated registration group. The results were visualized through scatter plots comparing each method with MsDCSwinT (<xref ref-type="fig" rid="F6">Figure 6A</xref>), box plots of mean CC distributions across methods (<xref ref-type="fig" rid="F6">Figure 6B</xref>), and scatter plots comparing each method with the classical ANTS method (<xref ref-type="fig" rid="F6">Figure 6C</xref>). The box plot analysis revealed a striking similarity in mean, quartiles, and outlier patterns between the proposed MsDCSwinT method and the classical ANTS method (<xref ref-type="fig" rid="F6">Figure 6B</xref>). Similarly, scatter plots with fitted regression lines (<xref ref-type="fig" rid="F6">Figure 6A</xref>, first panel) demonstrated a high degree of correlation between MsDCSwinT and ANTS, reinforcing the consistency of these findings. In contrast, comparatively lower similarity was observed between MsDCSwinT and the other three existing deep learning powered methods: C2FViT, C2FGALF, and ConvNet as shown in both scatter plots (<xref ref-type="fig" rid="F6">Figure 6A</xref>, second to fourth panels) and box plots (<xref ref-type="fig" rid="F6">Figure 6B</xref>). Interestingly, a similar pattern of reduced similarity was also observed when comparing ANTS with C2FViT, C2FGALF, and ConvNet (<xref ref-type="fig" rid="F6">Figures 6B</xref>, <xref ref-type="fig" rid="F6">C</xref>), suggesting consistent differentiation across methods.</p>
<fig position="float" id="F6">
<label>Figure 6</label>
<caption><p>Inter-group comparison of 159-seed functional connectivity across five automated registration methods in rats; <bold>(A)</bold> Scatter Plot Comparison with MsDCSwinT: The four panels illustrate pairwise comparisons of functional connectivity values between the proposed MsDCSwinT method and each of the other automated four methods: ANTS, C2FViT, C2FGALF, and ConvNet. Each scatter plot displays 1,256 correlation coefficient (CC) values with fitted linear regression lines (in red), representing the distribution and degree of correspondence between methods; <bold>(B)</bold> Box Plot of Mean Correlation Coefficients: This panel presents the distribution of mean CC values (<italic>n</italic> = 1,256) for each automated registration method including our MsDCSwinT. The box plots highlight the central tendency, interquartile range, and variability across methods, facilitating inter-method comparison of overall registration performance; <bold>(C)</bold> Scatter Plot Comparison with ANTs: Similar to <bold>(A)</bold>, these three panels display pairwise comparisons between classical ANTS identified as the best-performing baseline so far and the remaining three automated methods (C2FViT, C2FGALF, and ConvNet) based on 1,256 CC values each. Regression lines (in red) indicate linear trends in performance similarity.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1621244-g0006.tif">
<alt-text content-type="machine-generated">Panel A shows four scatter plots comparing MsDCSwinT with ANTS, C2VNT, C2FQALF, and ConvNet. Panel B displays a boxplot of correlation coefficients (CC) for these methods. Panel C presents scatter plots comparing ANTS with C2VNT, C2FQALF, and ConvNet, highlighting correlations with blue data points and red trend lines.</alt-text>
</graphic>
</fig>
<p>To further assess group-level differences, a pairwise t-test (<italic>p</italic> = 0.05, two-tailed, FDR-corrected, <italic>n</italic> = 8 per group) was conducted to compare the proposed MsDCSwinT method with each of the four other automated registration methods respectively across all 1,256 seed-to-seed correlation coefficient (CC) values (<xref ref-type="fig" rid="F7">Figure 7</xref>). Notably, very few statistically significant differences were observed between the MsDCSwinT method and the classical ANTS method (<xref ref-type="fig" rid="F7">Figure 7A</xref>), indicating a high degree of similarity in their resulting connectivity patterns. In contrast, a substantially greater number of connectivity differences were identified when MsDCSwinT was compared to the other three AI-powered methods: C2FViT, C2FGALF, and ConvNet (<xref ref-type="fig" rid="F7">Figure 7B</xref>), suggesting more variability in their alignment outcomes. To contextualize these results, additional comparisons between gold standard ANTS and each of the three AI methods were performed independently (<xref ref-type="fig" rid="F7">Figure 7C</xref>). Visual inspection suggests that the patterns of differences for C2FGALF and ConvNet were comparable to those observed in the MsDCSwinT vs. ANTs analysis (<xref ref-type="fig" rid="F7">Figures 7B</xref>, <xref ref-type="fig" rid="F7">C</xref>, middle and bottom panels), while ANTS appeared to outperform MsDCSwinT slightly in the comparison with C2FViT(<xref ref-type="fig" rid="F7">Figures 7B</xref>, <xref ref-type="fig" rid="F7">C</xref>, top panels). Collectively, these findings reinforce that MsDCSwinT achieves connectivity profiles closely aligned with those of the widely recognized ANTS method, supporting its reliability, robustness, and suitability as a fully automated solution for preprocessing in rodent resting-state fMRI studies. Consequently, the results underscore the potential of MsDCSwinT for fully automated and more accurate rs-fMRI preprocessing and analysis.</p>
<fig position="float" id="F7">
<label>Figure 7</label>
<caption><p>Inter-group statistical significance comparison of 159-seed functional connectivity across five automated registration methods in rats. Each panel illustrates the significant h values obtained by pairwise two-tailed <italic>t</italic>-test (<italic>p</italic> = 0.05, FDR-corrected, <italic>n</italic> = 8 per group), <bold>(A, B)</bold> comparing the proposed MsDCSwinT method to each of the four other automated registration methods, <bold>(C)</bold> comparing the gold standard ANTS to each of the three other automated AI registration methods across all 1,256 seed-to-seed correlation coefficient (CC) values. Statistically significant differences between node pair are represented using parula color bar with each value 1; <bold>(A)</bold> MsDCSwinT vs. ANTS; <bold>(B)</bold> top: MsDCSwinT vs. C2FViT, middle: MsDCSwinT vs. C2FGALF, bottom: MsDCSwinT vs. C2FGALF; <bold>(C)</bold> top: ANTS vs. C2FViT, middle: ANTS vs. C2FGALF, bottom: ANTS vs. C2FGALF; This analysis highlights method-specific statistically significant variations in alignment performance, as reflected in functional connectivity outcomes derived from atlas-based seed-to-seed correlation analysis.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-19-1621244-g0007.tif">
<alt-text content-type="machine-generated">Six circular graphs labeled A to C, comparing MsDCSwinT and ANTS algorithms against C2FViT, C2FGALF, and ConvNet. Each graph shows networks of colored lines, ranging from purple to yellow. A color bar at the bottom represents h values from 0.1 to 1.</alt-text>
</graphic>
</fig>
</sec></sec></sec>
<sec sec-type="discussion" id="s4">
<label>4</label>
<title>Discussion</title>
<sec>
<label>4.1</label>
<title>Affine registration analysis</title>
<p>The results show that our method performs similarly to the traditional method (ANTS), and outperforms the learning-based methods like ConvNet, C2FViT, and C2FGALF. Importantly, our method improves atlas-based registration on both CTNI fMRI datasets. Since dataset 2 was not used during training, the strong performance on it demonstrates that our method generalizes well to unseen fMRI data. Our proposed MsDCSwinT method outperforms both C2FViT and C2FGALF in terms of registration accuracy and inference speed. C2FViT introduced a coarse-to-fine vision transformer architecture to model long-range dependencies for affine registration but showed limited capability in effectively capturing multi-scale features. C2FGALF improved upon this by using multiscale convolutional kernels and a weighted global positional attention mechanism to better fuse global and local feature mappings. However, C2FGALF still faces challenges in optimally balancing fine-scale and global-scale information. In contrast, our method integrates dilated convolutions, allowing MsDCSwinT to capture local anatomical features while the transformer layers handle the global alignment. The combination of dilated convolutions and Swin Transformers in a multi-stage framework ensures that both local and global misalignments are addressed progressively, which is essential for accurate registration in fMRI data. The multi-stage approach and the Swin Transformer&#x00027;s ability to handle multi-scale feature extraction further improve its robustness compared to C2FViT and C2FGALF, making MsDCSwinT a more effective solution for the complex challenges of preclinical fMRI registration.</p>
<p>The proposed method not only matches the registration accuracy of traditional algorithm, ANTS, but also offers a substantial improvement in average inference time. Specifically, our affine model achieves significantly faster inference compared to traditional method (ANTS) and performs comparably to learning-based affine registration approaches such as ConvNet, C2FViT, and C2FGALF.</p>
<p>Our method uses the global connectivity of the self-attention operator while controlling the locality of the convolutional feed-forward layer. This allows it to capture global orientations, spatial positions, and long-term dependencies between the image pair to compute a set of geometric transformation parameters. Extensive experiments show that our method outperforms existing techniques, and is robust when tested on unseen dataset. It also performs slightly better than conventional method ANTS in term of registration accuracy and maintains the runtime advantages of learning-based approaches.</p>
<sec>
<label>4.1.1</label>
<title>Limitations</title>
<p>Despite its promising results, our proposed method has several limitations. It is currently tailored for affine registration and has not been extended to deformable registration tasks. Furthermore, this study was conducted using data acquired from a single institute with consistent imaging parameters. Because MRI contrast is highly dependent on scanner settings such as TR and TE, the generalizability of MsDCSwinT to data acquired using different MRI sequences or substantially different scan parameters has not yet been evaluated. In practical applications, additional training or fine-tuning may be required when deploying the model across sites with different acquisition protocols. Future work will include multi-center validation to assess performance across varied scanning environments. In this study, the test Dataset 1 was selected from the training dataset using cross-validation, while Dataset 2 comprised independent subjects from other studies and experimental conditions. This design supports intra-center generalization, and future work will focus on inter-center validation to evaluate robustness across imaging platforms and acquisition protocols. We also did not include comparisons with manual registration due to its time-consuming nature and reliance on expert interpretation. Nevertheless, incorporating such comparisons in future work could provide valuable insights into the method&#x00027;s performance in preclinical imaging studies.</p>
</sec>
</sec>
<sec>
<label>4.2</label>
<title>Functional connectivity analysis</title>
<p>Seed-to-seed functional connectivity (FC) analysis remains a cornerstone in neuroimaging research, offering an atlas-based, hypothesis-driven framework to probe whole-brain functional organization. By utilizing predefined anatomical regions of interest (ROIs), seed-based FC decomposes complex neuroimaging data into spatially meaningful units, allowing researchers to quantify interregional communication and network-level brain function (<xref ref-type="bibr" rid="B30">Lupinsky et al., 2025</xref>; <xref ref-type="bibr" rid="B43">Sourty et al., 2024a</xref>,<xref ref-type="bibr" rid="B44">b</xref>; <xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>; <xref ref-type="bibr" rid="B28">Karatas et al., 2021</xref>; <xref ref-type="bibr" rid="B38">Nasseef et al., 2019</xref>; <xref ref-type="bibr" rid="B3">Boulos et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Hamida et al., 2018</xref>; <xref ref-type="bibr" rid="B5">Charbogne et al., 2017</xref>; <xref ref-type="bibr" rid="B35">Nasseef, 2015</xref>). Moreover, this approach is particularly powerful for assessing the influence of preprocessing steps such as spatial normalization and registration on downstream connectivity outcomes. In the present study, we combined a high-resolution, down-sampled anatomical rat brain atlas with a seed-to-seed FC framework to systematically evaluate the impact of multiple automated registration methods including our proposed unsupervised model MsDCSwinT against established techniques such as ANTS and alternative deep learning models (C2FViT, C2FGALF, and ConvNet).</p>
<p>Our findings demonstrate that the proposed MsDCSwinT method exhibits strong concordance with the classical ANTS algorithm, widely regarded as a gold standard for non-linear registration, while offering significant improvements in computational efficiency. Scatter plot comparisons (<xref ref-type="fig" rid="F6">Figures 6A</xref>, <xref ref-type="fig" rid="F6">C</xref>) and box plot analyses (<xref ref-type="fig" rid="F6">Figure 6B</xref>) reveal that although the overall distributions of correlation coefficients (CC) are similar across methods, subtle yet meaningful variations in CC values suggest method-specific influences on alignment precision. These differences become more evident in pairwise group comparisons at the subject level, as assessed by two-sample t-tests (<italic>p</italic> = 0.05, FDR-corrected, <italic>n</italic> = 8 per group) (<xref ref-type="fig" rid="F7">Figure 7</xref>). Importantly, the presence of these small but statistically discernible variations highlights the necessity of employing multimodal evaluation strategies during preprocessing validation, as reliance on single similarity metrics may obscure critical nuances (<xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>, <xref ref-type="bibr" rid="B38">2019</xref>; <xref ref-type="bibr" rid="B3">Boulos et al., 2019</xref>)). By integrating both statistical and visualization-based analyses (<xref ref-type="fig" rid="F6">Figures 6</xref>, <xref ref-type="fig" rid="F7">7</xref>), our study underscores the need to benchmark AI-driven registration pipelines not only in terms of algorithmic accuracy but also based on their downstream impact on functional connectivity outcomes. Collectively, these results establish MsDCSwinT as a robust, scalable, and reliable preprocessing tool for rodent fMRI research, with potential to enhance reproducibility and methodological rigor in preclinical neuroimaging workflows.</p>
<sec>
<label>4.2.1</label>
<title>Limitations</title>
<p>For simplicity and clarity, we have taken the advantage of using h-values from the pairwise <italic>t</italic>-tests (<italic>p</italic> = 0.05, two-tailed, FDR-corrected, <italic>n</italic> = 8 per group) to highlight statistically significant differences in functional connectivity (<xref ref-type="fig" rid="F7">Figure 7</xref>). Importantly, our primary aim was to identify the presence or absence of significant differences, rather than to assess the magnitude or strength of those differences in the form of p-values or tstat-values, which are more commonly interpreted in the context of biological or functional meaning (<xref ref-type="bibr" rid="B30">Lupinsky et al., 2025</xref>; <xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>; <xref ref-type="bibr" rid="B28">Karatas et al., 2021</xref>; <xref ref-type="bibr" rid="B38">Nasseef et al., 2019</xref>; <xref ref-type="bibr" rid="B3">Boulos et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Hamida et al., 2018</xref>; <xref ref-type="bibr" rid="B5">Charbogne et al., 2017</xref>). In this context, the binary nature of h-values (1 = significant vs. 0 = not significant) effectively serves our objective of comparing the different registration methods from a methodological perspective. Therefore, we deliberately chose not to present the associated p-values, as our interest lies in the existence of statistically significant differences, not in their relative quantification (<xref ref-type="bibr" rid="B30">Lupinsky et al., 2025</xref>; <xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>, <xref ref-type="bibr" rid="B38">2019</xref>; <xref ref-type="bibr" rid="B3">Boulos et al., 2019</xref>) or degree (<xref ref-type="bibr" rid="B46">Wang et al., 2025</xref>; <xref ref-type="bibr" rid="B15">Fuini et al., 2025</xref>) or effect size (<xref ref-type="bibr" rid="B36">Nasseef et al., 2018</xref>) or dynamic functional connectivity (<xref ref-type="bibr" rid="B43">Sourty et al., 2024a</xref>,<xref ref-type="bibr" rid="B44">b</xref>). Additionally, we did not perform further analysis such as hypothesis free independent component analysis (ICA) (<xref ref-type="bibr" rid="B30">Lupinsky et al., 2025</xref>; <xref ref-type="bibr" rid="B37">Nasseef et al., 2021</xref>; <xref ref-type="bibr" rid="B28">Karatas et al., 2021</xref>; <xref ref-type="bibr" rid="B38">Nasseef et al., 2019</xref>; <xref ref-type="bibr" rid="B18">Hamida et al., 2018</xref>; <xref ref-type="bibr" rid="B5">Charbogne et al., 2017</xref>) or directed (model based/model free) functional connectivity (<xref ref-type="bibr" rid="B18">Hamida et al., 2018</xref>; <xref ref-type="bibr" rid="B5">Charbogne et al., 2017</xref>) as these methods fall outside the primary scope of our current investigation. Given that our objective was to find the benchmark automated registration methods using atlas-based seed-to-seed correlation analysis, additional exploratory or directional techniques were deemed unnecessary. Equally, these alternative approaches are unlikely to provide further insights relevant to our methodological comparison and may instead introduce confounding variability unrelated to the core aim of evaluating registration performance.</p></sec></sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusion and future work</title>
<p>In this work, we presented a novel fMRI processing pipeline that combines a GAN-based denoising approach, transformer-based skull stripping, and affine registration methods. The proposed pipeline significantly enhances fMRI data preprocessing, with a main focus on registration in this study. It demonstrates superior performance in both registration accuracy and computational efficiency compared to traditional and learning-based methods. By integrating advanced deep learning models, such as GANs and transformers, the pipeline enables robust and automated processing, even in the absence of labelled data. In this paper, the primary emphasis was on the registration component. In future work, we plan to extend our evaluation by comparing the overall pipeline against other established fMRI preprocessing pipelines to further validate its effectiveness. Moreover, while we currently employ AFNI motion correction, we aim to develop our own motion correction method tailored to our specific pipeline. Motion artifacts are a common challenge in fMRI data, and we aim to develop strategies for reducing these artifacts to further improve the quality of processed fMRI data. Additionally, we plan to extend our work to incorporate deformable registration, allowing for more flexible alignment between images with complex transformations. Furthermore, we intend to test and optimize our pipeline on a wider variety of fMRI datasets, which will help evaluate its generalizability and applicability across different populations and experimental conditions. These extensions will enable our pipeline to become a more powerful and versatile tool for fMRI analysis, with potential applications in preclinical and research settings.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>All animal procedures followed the Guide for the Care and Use of Laboratory Animals (NIH Publication No. 85-23, Revised 1985) and were approved by the Institutional Animal Care and Use Committee at Northeastern University, adhering to NIH and AALAS guidelines. The study was conducted in accordance with the local legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>SS: Formal analysis, Investigation, Methodology, Project administration, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. MN: Formal analysis, Investigation, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. RU: Formal analysis, Investigation, Software, Writing &#x02013; review &#x00026; editing. AC: Data curation, Software, Writing &#x02013; review &#x00026; editing. DM: Funding acquisition, Project administration, Resources, Supervision, Writing &#x02013; review &#x00026; editing. PK: Data curation, Project administration, Resources, Supervision, Writing &#x02013; review &#x00026; editing. CF: Funding acquisition, Project administration, Resources, Supervision, Writing &#x02013; review &#x00026; editing. CJ: Funding acquisition, Project administration, Resources, Supervision, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<ack><title>Acknowledgments</title><p>We would like to acknowledge the support of the Tessellis Ltd.</p></ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>DM was employed by company Tessellis Ltd. DM has financial interest in Tessellis Ltd.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The authors declare that this study received funding from Tessellis Ltd. and Mitacs Inc (grant &#x00023;: IT40950). DM has financial interest in Tessellis, and had the following involvement in the study: design, planning and data acquisition.</p>
</sec>
<sec sec-type="ai-statement" id="s10">
<title>Generative AI statement</title>
<p>The author(s) declare that no Gen AI was used in the creation of this manuscript. Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p></sec>
<sec sec-type="disclaimer" id="s11">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Anderson</surname> <given-names>R. J.</given-names></name> <name><surname>Cook</surname> <given-names>J. J.</given-names></name> <name><surname>Delpratt</surname> <given-names>N.</given-names></name> <name><surname>Nouls</surname> <given-names>J. C.</given-names></name> <name><surname>Gu</surname> <given-names>B.</given-names></name> <name><surname>McNamara</surname> <given-names>J. O.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Small animal multivariate brain analysis (samba)-a high throughput pipeline with a validation framework</article-title>. <source>Neuroinformatics</source> <volume>17</volume>, <fpage>451</fpage>&#x02013;<lpage>472</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s12021-018-9410-0</pub-id></mixed-citation>
</ref>
<ref id="B2">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Avants</surname> <given-names>B. B.</given-names></name> <name><surname>Tustison</surname> <given-names>N. J.</given-names></name> <name><surname>Song</surname> <given-names>G.</given-names></name> <name><surname>Cook</surname> <given-names>P. A.</given-names></name> <name><surname>Klein</surname> <given-names>A.</given-names></name> <name><surname>Gee</surname> <given-names>J. C.</given-names></name></person-group> (<year>2011</year>). <article-title>A reproducible evaluation of ants similarity metric performance in brain image registration</article-title>. <source>Neuroimage</source> <volume>54</volume>, <fpage>2033</fpage>&#x02013;<lpage>2044</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2010.09.025</pub-id><pub-id pub-id-type="pmid">20851191</pub-id></mixed-citation>
</ref>
<ref id="B3">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boulos</surname> <given-names>L.-J.</given-names></name> <name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>McNicholas</surname> <given-names>M.</given-names></name> <name><surname>Mechling</surname> <given-names>A.</given-names></name> <name><surname>Harsan</surname> <given-names>L. A.</given-names></name> <name><surname>Darcq</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Touchscreen-based phenotyping: altered stimulus/reward association and lower perseveration to gain a reward in mu opioid receptor knockout mice</article-title>. <source>Sci. Rep</source>. <volume>9</volume>:<fpage>4044</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-019-40622-6</pub-id><pub-id pub-id-type="pmid">30858487</pub-id></mixed-citation>
</ref>
<ref id="B4">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bricq</surname> <given-names>S.</given-names></name> <name><surname>Kidane</surname> <given-names>H. L.</given-names></name> <name><surname>Zavala-Bojorquez</surname> <given-names>J.</given-names></name> <name><surname>Oudot</surname> <given-names>A.</given-names></name> <name><surname>Vrigneaud</surname> <given-names>J.-M.</given-names></name> <name><surname>Brunotte</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Automatic deformable pet/mri registration for preclinical studies based on b-splines and non-linear intensity transformation</article-title>. <source>Med. Biol. Eng. Comput</source>. <volume>56</volume>:<fpage>1531</fpage>&#x02013;<lpage>1539</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11517-018-1797-0</pub-id><pub-id pub-id-type="pmid">29411247</pub-id></mixed-citation>
</ref>
<ref id="B5">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Charbogne</surname> <given-names>P.</given-names></name> <name><surname>Gardon</surname> <given-names>O.</given-names></name> <name><surname>Mart&#x000ED;n-Garc&#x000ED;a</surname> <given-names>E.</given-names></name> <name><surname>Keyworth</surname> <given-names>H. L.</given-names></name> <name><surname>Matsui</surname> <given-names>A.</given-names></name> <name><surname>Mechling</surname> <given-names>A. E.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Mu opioid receptors in gamma-aminobutyric acidergic forebrain neurons moderate motivation for heroin and palatable food</article-title>. <source>Biol. Psychiatry</source> <volume>81</volume>, <fpage>778</fpage>&#x02013;<lpage>788</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.biopsych.2016.12.022</pub-id><pub-id pub-id-type="pmid">28185645</pub-id></mixed-citation>
</ref>
<ref id="B6">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Diaz-Pinto</surname> <given-names>A.</given-names></name> <name><surname>Ravikumar</surname> <given-names>N.</given-names></name> <name><surname>Frangi</surname> <given-names>A. F.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning in medical image registration</article-title>. <source>Progr. Biomed. Eng</source>. <volume>3</volume>:<fpage>012003</fpage>. doi: <pub-id pub-id-type="doi">10.1088/2516-1091/abd37c</pub-id></mixed-citation>
</ref>
<ref id="B7">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>K.</given-names></name> <name><surname>Fung</surname> <given-names>K.-M.</given-names></name> <name><surname>Thai</surname> <given-names>T. C.</given-names></name> <name><surname>Moore</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Recent advances and clinical applications of deep learning in medical image analysis</article-title>. <source>Med. Image Anal</source>. <volume>79</volume>:<fpage>102444</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.media.2022.102444</pub-id><pub-id pub-id-type="pmid">35472844</pub-id></mixed-citation>
</ref>
<ref id="B8">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cox</surname> <given-names>R. W.</given-names></name></person-group> (<year>1996</year>). <article-title>Afni: software for analysis and visualization of functional magnetic resonance neuroimages</article-title>. <source>Comput. Biomed. Res</source>. <volume>29</volume>, <fpage>162</fpage>&#x02013;<lpage>173</lpage>. doi: <pub-id pub-id-type="doi">10.1006/cbmr.1996.0014</pub-id><pub-id pub-id-type="pmid">8812068</pub-id></mixed-citation>
</ref>
<ref id="B9">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>De Vos</surname> <given-names>B. D.</given-names></name> <name><surname>Berendsen</surname> <given-names>F. F.</given-names></name> <name><surname>Viergever</surname> <given-names>M. A.</given-names></name> <name><surname>Sokooti</surname> <given-names>H.</given-names></name> <name><surname>Staring</surname> <given-names>M.</given-names></name> <name><surname>I&#x00161;gum</surname> <given-names>I.</given-names></name></person-group> (<year>2019</year>). <article-title>A deep learning framework for unsupervised affine and deformable image registration</article-title>. <source>Med. Image Anal</source>. <volume>52</volume>, <fpage>128</fpage>&#x02013;<lpage>143</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.media.2018.11.010</pub-id><pub-id pub-id-type="pmid">30579222</pub-id></mixed-citation>
</ref>
<ref id="B10">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Di</surname> <given-names>X.</given-names></name> <name><surname>Biswal</surname> <given-names>B. B.</given-names></name></person-group> (<year>2023</year>). <article-title>A functional mri pre-processing and quality control protocol based on statistical parametric mapping (spm) and matlab</article-title>. <source>Front. Neuroimaging</source> <volume>1</volume>:<fpage>1070151</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnimg.2022.1070151</pub-id><pub-id pub-id-type="pmid">37555150</pub-id></mixed-citation>
</ref>
<ref id="B11">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dice</surname> <given-names>L. R.</given-names></name></person-group> (<year>1945</year>). <article-title>Measures of the amount of ecologic association between species</article-title>. <source>Ecology</source> <volume>26</volume>, <fpage>297</fpage>&#x02013;<lpage>302</lpage>. doi: <pub-id pub-id-type="doi">10.2307/1932409</pub-id></mixed-citation>
</ref>
<ref id="B12">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A.</given-names></name> <name><surname>Beyer</surname> <given-names>L.</given-names></name> <name><surname>Kolesnikov</surname> <given-names>A.</given-names></name> <name><surname>Weissenborn</surname> <given-names>D.</given-names></name> <name><surname>Zhai</surname> <given-names>X.</given-names></name> <name><surname>Unterthiner</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>An image is worth 16x16 words: transformers for image recognition at scale</article-title>. <source>arXiv preprint arXiv:2010.11929</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id></mixed-citation>
</ref>
<ref id="B13">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Esteban</surname> <given-names>O.</given-names></name> <name><surname>Markiewicz</surname> <given-names>C. J.</given-names></name> <name><surname>Blair</surname> <given-names>R. W.</given-names></name> <name><surname>Moodie</surname> <given-names>C. A.</given-names></name> <name><surname>Isik</surname> <given-names>A. I.</given-names></name> <name><surname>Erramuzpe</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>fmriprep: a robust preprocessing pipeline for functional mri</article-title>. <source>Nat. Methods</source> <volume>16</volume>, <fpage>111</fpage>&#x02013;<lpage>116</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41592-018-0235-4</pub-id><pub-id pub-id-type="pmid">30532080</pub-id></mixed-citation>
</ref>
<ref id="B14">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>L.</given-names></name> <name><surname>Chen</surname> <given-names>Y.</given-names></name> <name><surname>Ji</surname> <given-names>W.</given-names></name> <name><surname>Yang</surname> <given-names>F.</given-names></name></person-group> (<year>2024</year>). <article-title>Sstrans-net: smart swin transformer network for medical image segmentation</article-title>. <source>Biomed. Signal Process. Control</source> <volume>91</volume>:<fpage>106071</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bspc.2024.106071</pub-id></mixed-citation>
</ref>
<ref id="B15">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fuini</surname> <given-names>E.</given-names></name> <name><surname>Chang</surname> <given-names>A.</given-names></name> <name><surname>Ortiz</surname> <given-names>R. J.</given-names></name> <name><surname>Nasseef</surname> <given-names>T.</given-names></name> <name><surname>Edwards</surname> <given-names>J.</given-names></name> <name><surname>Latta</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Dose-dependent changes in global brain activity and functional connectivity following exposure to psilocybin: a bold mri study in awake rats</article-title>. <source>bioRxiv</source>. doi: <pub-id pub-id-type="doi">10.3389/fnins.2025.1554049</pub-id><pub-id pub-id-type="pmid">40376612</pub-id></mixed-citation>
</ref>
<ref id="B16">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>J.</given-names></name> <name><surname>Gong</surname> <given-names>M.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name></person-group> (<year>2022</year>). <article-title>Congested crowd instance localization with dilated convolutional swin transformer</article-title>. <source>Neurocomputing</source> <volume>513</volume>, <fpage>94</fpage>&#x02013;<lpage>103</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neucom.2022.09.113</pub-id></mixed-citation>
</ref>
<ref id="B17">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Golestani</surname> <given-names>N.</given-names></name> <name><surname>Wang</surname> <given-names>A.</given-names></name> <name><surname>Moallem</surname> <given-names>G.</given-names></name> <name><surname>Bean</surname> <given-names>G. R.</given-names></name> <name><surname>Rusu</surname> <given-names>M.</given-names></name></person-group> (<year>2025</year>). <article-title>Pvit-air: puzzling vision transformer-based affine image registration for multi histopathology and faxitron images of breast tissue</article-title>. <source>Med. Image Anal</source>. <volume>99</volume>:<fpage>103356</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.media.2024.103356</pub-id><pub-id pub-id-type="pmid">39378568</pub-id></mixed-citation>
</ref>
<ref id="B18">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hamida</surname> <given-names>S. B.</given-names></name> <name><surname>Mendon&#x000E7;a-Netto</surname> <given-names>S.</given-names></name> <name><surname>Arefin</surname> <given-names>T. M.</given-names></name> <name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Boulos</surname> <given-names>L.-J.</given-names></name> <name><surname>McNicholas</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Increased alcohol seeking in mice lacking gpr88 involves dysfunctional mesocorticolimbic networks</article-title>. <source>Biol. Psychiatry</source> <volume>84</volume>, <fpage>202</fpage>&#x02013;<lpage>212</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.biopsych.2018.01.026</pub-id><pub-id pub-id-type="pmid">29580570</pub-id></mixed-citation>
</ref>
<ref id="B19">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hasani</surname> <given-names>S. J.</given-names></name> <name><surname>Rakhshanpour</surname> <given-names>A.</given-names></name> <name><surname>Tehrani</surname> <given-names>A.-A.</given-names></name> <name><surname>Enferadi</surname> <given-names>A.</given-names></name></person-group> (<year>2025</year>). <article-title>A review article on diagnostic imaging applications in the diagnosis of infectious diseases in small animals</article-title>. <source>Front. Biomed. Technol</source>. 12.</mixed-citation>
</ref>
<ref id="B20">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>W.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>I.</given-names></name> <name><surname>Wang</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>A coarse-to-fine deformable transformation framework for unsupervised multi-contrast mr image registration with dual consistency constraint</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>40</volume>, <fpage>2589</fpage>&#x02013;<lpage>2599</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2021.3059282</pub-id><pub-id pub-id-type="pmid">33577451</pub-id></mixed-citation>
</ref>
<ref id="B21">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huttenlocher</surname> <given-names>D. P.</given-names></name> <name><surname>Klanderman</surname> <given-names>G. A.</given-names></name> <name><surname>Rucklidge</surname> <given-names>W. J.</given-names></name></person-group> (<year>1993</year>). <article-title>Comparing images using the hausdorff distance</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>15</volume>, <fpage>850</fpage>&#x02013;<lpage>863</lpage>. doi: <pub-id pub-id-type="doi">10.1109/34.232073</pub-id></mixed-citation>
</ref>
<ref id="B22">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Iglesias</surname> <given-names>J. E.</given-names></name></person-group> (<year>2023</year>). <article-title>A ready-to-use machine learning tool for symmetric multi-modality registration of brain mri</article-title>. <source>Sci. Rep</source>. <volume>13</volume>:<fpage>6657</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41598-023-33781-0</pub-id><pub-id pub-id-type="pmid">37095168</pub-id></mixed-citation>
</ref>
<ref id="B23">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ioanas</surname> <given-names>H.-I.</given-names></name> <name><surname>Marks</surname> <given-names>M.</given-names></name> <name><surname>Zerbi</surname> <given-names>V.</given-names></name> <name><surname>Yanik</surname> <given-names>M. F.</given-names></name> <name><surname>Rudin</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>An optimized registration workflow and standard geometric space for small animal brain imaging</article-title>. <source>NeuroImage</source> <volume>241</volume>:<fpage>118386</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2021.118386</pub-id><pub-id pub-id-type="pmid">34280528</pub-id></mixed-citation>
</ref>
<ref id="B24">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Ioffe</surname> <given-names>S.</given-names></name> <name><surname>Szegedy</surname> <given-names>C.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Batch normalization: accelerating deep network training by reducing internal covariate shift,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>PMLR</publisher-loc>), <fpage>448</fpage>&#x02013;<lpage>456</lpage>.</mixed-citation>
</ref>
<ref id="B25">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jaderberg</surname> <given-names>M.</given-names></name> <name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name> <name><surname>Kavukcuoglu</surname> <given-names>K.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Spatial transformer networks,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems, Vol. 28</source>.</mixed-citation>
</ref>
<ref id="B26">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jenkinson</surname> <given-names>M.</given-names></name> <name><surname>Beckmann</surname> <given-names>C. F.</given-names></name> <name><surname>Behrens</surname> <given-names>T. E.</given-names></name> <name><surname>Woolrich</surname> <given-names>M. W.</given-names></name> <name><surname>Smith</surname> <given-names>S. M.</given-names></name></person-group> (<year>2012</year>). <article-title>Fsl</article-title>. <source>Neuroimage</source> <volume>62</volume>, <fpage>782</fpage>&#x02013;<lpage>790</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuroimage.2011.09.015</pub-id><pub-id pub-id-type="pmid">21979382</pub-id></mixed-citation>
</ref>
<ref id="B27">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ji</surname> <given-names>W.</given-names></name> <name><surname>Yang</surname> <given-names>F.</given-names></name></person-group> (<year>2024</year>). <article-title>Affine medical image registration with fusion feature mapping in local and global</article-title>. <source>Phys. Medi. Biol</source>. <volume>69</volume>:<fpage>055029</fpage>. doi: <pub-id pub-id-type="doi">10.1088/1361-6560/ad2717</pub-id><pub-id pub-id-type="pmid">38324893</pub-id></mixed-citation>
</ref>
<ref id="B28">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Karatas</surname> <given-names>M.</given-names></name> <name><surname>Noblet</surname> <given-names>V.</given-names></name> <name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Bienert</surname> <given-names>T.</given-names></name> <name><surname>Reisert</surname> <given-names>M.</given-names></name> <name><surname>Hennig</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Mapping the living mouse brain neural architecture: strain-specific patterns of brain structural and functional connectivity</article-title>. <source>Brain Struct. Funct</source>. <volume>226</volume>, <fpage>647</fpage>&#x02013;<lpage>669</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s00429-020-02190-8</pub-id><pub-id pub-id-type="pmid">33635426</pub-id></mixed-citation>
</ref>
<ref id="B29">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name> <name><surname>Cao</surname> <given-names>Y.</given-names></name> <name><surname>Hu</surname> <given-names>H.</given-names></name> <name><surname>Wei</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Swin transformer: hierarchical vision transformer using shifted windows,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF International Conference on Computer Vision</source>, 10012&#x02013;10022. doi: <pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00986</pub-id></mixed-citation>
</ref>
<ref id="B30">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lupinsky</surname> <given-names>D.</given-names></name> <name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Parent</surname> <given-names>C.</given-names></name> <name><surname>Craig</surname> <given-names>K.</given-names></name> <name><surname>Diorio</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>T.-Y.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Resting-state fMRI reveals altered functional connectivity associated with resilience and susceptibility to chronic social defeat stress in mouse brain</article-title>. <source>Mol. Psychiatry</source> <volume>30</volume>, <fpage>2943</fpage>&#x02013;<lpage>2954</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41380-025-02897-2</pub-id><pub-id pub-id-type="pmid">39984680</pub-id></mixed-citation>
</ref>
<ref id="B31">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mok</surname> <given-names>T. C.</given-names></name> <name><surname>Chung</surname> <given-names>A.</given-names></name></person-group> (<year>2020a</year>). <article-title>&#x0201C;Fast symmetric diffeomorphic image registration with convolutional neural networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 4644&#x02013;4653. doi: <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.00470</pub-id></mixed-citation>
</ref>
<ref id="B32">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mok</surname> <given-names>T. C.</given-names></name> <name><surname>Chung</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Affine medical image registration with coarse-to-fine vision transformer,&#x0201D;</article-title> in <source>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</source>, 20835&#x02013;20844. doi: <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.02017</pub-id></mixed-citation>
</ref>
<ref id="B33">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Mok</surname> <given-names>T. C.</given-names></name> <name><surname>Chung</surname> <given-names>A. C.</given-names></name></person-group> (<year>2020b</year>). <article-title>&#x0201C;Large deformation diffeomorphic image registration with laplacian pyramid networks,&#x0201D;</article-title> in <source>Medical Image Computing and Computer Assisted Intervention-MICCAI 2020: 23rd International Conference, Lima, Peru, October 4-8, 2020, Proceedings, Part III 23</source> (<publisher-loc>Springer</publisher-loc>), <fpage>211</fpage>&#x02013;<lpage>221</lpage>. doi: <pub-id pub-id-type="doi">10.1007/978-3-030-59716-0_21</pub-id></mixed-citation>
</ref>
<ref id="B34">
<mixed-citation publication-type="book"><person-group person-group-type="author"><name><surname>Mok</surname> <given-names>T. C.</given-names></name> <name><surname>Chung</surname> <given-names>A. C.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Conditional deformable image registration with convolutional neural network,&#x0201D;</article-title> in <source>Medical Image Computing and Computer Assisted Intervention-MICCAI 2021: 24th International Conference, Strasbourg, France, September 27-October 1, 2021, Proceedings, Part IV 24</source> (<publisher-loc>Springer</publisher-loc>), <fpage>35</fpage>&#x02013;<lpage>45</lpage>. doi: <pub-id pub-id-type="doi">10.1007/978-3-030-87202-1_4</pub-id></mixed-citation>
</ref>
<ref id="B35">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nasseef</surname> <given-names>M. T.</given-names></name></person-group> (<year>2015</year>). <source>Measuring Directed Functional Connectivity in Mouse FMRI Networks Using Granger Causality</source>.</mixed-citation>
</ref>
<ref id="B36">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Devenyi</surname> <given-names>G. A.</given-names></name> <name><surname>Mechling</surname> <given-names>A. E.</given-names></name> <name><surname>Harsan</surname> <given-names>L.-A.</given-names></name> <name><surname>Chakravarty</surname> <given-names>M. M.</given-names></name> <name><surname>Kieffer</surname> <given-names>B. L.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Deformation-based morphometry mri reveals brain structural modifications in living mu opioid receptor knockout mice</article-title>. <source>Front. Psychiatry</source> <volume>9</volume>:<fpage>643</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fpsyt.2018.00643</pub-id><pub-id pub-id-type="pmid">30559685</pub-id></mixed-citation>
</ref>
<ref id="B37">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Ma</surname> <given-names>W.</given-names></name> <name><surname>Singh</surname> <given-names>J. P.</given-names></name> <name><surname>Dozono</surname> <given-names>N.</given-names></name> <name><surname>Lan&#x000E7;on</surname> <given-names>K.</given-names></name> <name><surname>S&#x000E9;gu&#x000E9;la</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Chronic generalized pain disrupts whole brain functional connectivity in mice</article-title>. <source>Brain Imaging Behav.</source> <volume>15</volume>, <fpage>2406</fpage>&#x02013;<lpage>2416</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11682-020-00438-9</pub-id><pub-id pub-id-type="pmid">33428113</pub-id></mixed-citation>
</ref>
<ref id="B38">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Singh</surname> <given-names>J. P.</given-names></name> <name><surname>Ehrlich</surname> <given-names>A. T.</given-names></name> <name><surname>McNicholas</surname> <given-names>M.</given-names></name> <name><surname>Park</surname> <given-names>D. W.</given-names></name> <name><surname>Ma</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Oxycodone-mediated activation of the mu opioid receptor reduces whole brain functional connectivity in mice</article-title>. <source>ACS Pharmacol. Transl. Sci</source>. <volume>2</volume>, <fpage>264</fpage>&#x02013;<lpage>274</lpage>. doi: <pub-id pub-id-type="doi">10.1021/acsptsci.9b00021</pub-id><pub-id pub-id-type="pmid">32259060</pub-id></mixed-citation>
</ref>
<ref id="B39">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nieto-Castanon</surname> <given-names>A.</given-names></name></person-group> (<year>2022</year>). <article-title>Preparing fmri data for statistical analysis</article-title>. <source>arXiv preprint arXiv:2210.13564</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2210.13564</pub-id></mixed-citation>
</ref>
<ref id="B40">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>W.</given-names></name> <name><surname>Ji</surname> <given-names>B.</given-names></name> <name><surname>Guan</surname> <given-names>Y.</given-names></name> <name><surname>Cao</surname> <given-names>L.</given-names></name> <name><surname>Ni</surname> <given-names>R.</given-names></name></person-group> (<year>2022</year>). <article-title>Recent technical advances in accelerating the clinical translation of small animal brain imaging: hybrid imaging, deep learning, and transcriptomics</article-title>. <source>Front. Med</source>. <volume>9</volume>:<fpage>771982</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fmed.2022.771982</pub-id><pub-id pub-id-type="pmid">35402436</pub-id></mixed-citation>
</ref>
<ref id="B41">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Soltanpour</surname> <given-names>S.</given-names></name> <name><surname>Chang</surname> <given-names>A.</given-names></name> <name><surname>Madularu</surname> <given-names>D.</given-names></name> <name><surname>Kulkarni</surname> <given-names>P.</given-names></name> <name><surname>Ferris</surname> <given-names>C.</given-names></name> <name><surname>Joslin</surname> <given-names>C.</given-names></name></person-group> (<year>2025a</year>). 3d wasserstein generative adversarial network with dense u-net-based discriminator for preclinical fmri denoising. <italic>J. Imaging Inform. Med</italic>. doi: <pub-id pub-id-type="doi">10.1007/s10278-025-01434-5.</pub-id> [Epub ahead of print].</mixed-citation>
</ref>
<ref id="B42">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Soltanpour</surname> <given-names>S.</given-names></name> <name><surname>Utama</surname> <given-names>R.</given-names></name> <name><surname>Chang</surname> <given-names>A.</given-names></name> <name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Madularu</surname> <given-names>D.</given-names></name> <name><surname>Kulkarni</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2025b</year>). <article-title>Sst-dunet: smart swin transformer and dense unet for automated preclinical fmri skull stripping</article-title>. <source>J. Neurosci. Methods</source> <volume>423</volume>:<fpage>110545</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jneumeth.2025.110545</pub-id><pub-id pub-id-type="pmid">40789440</pub-id></mixed-citation>
</ref>
<ref id="B43">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sourty</surname> <given-names>M.</given-names></name> <name><surname>Champagnol-Di Liberti</surname> <given-names>C.</given-names></name> <name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Welsch</surname> <given-names>L.</given-names></name> <name><surname>Noblet</surname> <given-names>V.</given-names></name> <name><surname>Darcq</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2024a</year>). <article-title>Chronic morphine leaves a durable fingerprint on whole-brain functional connectivity</article-title>. <source>Biol. Psychiatry</source> <volume>96</volume>, <fpage>708</fpage>&#x02013;<lpage>716</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.biopsych.2023.12.007</pub-id><pub-id pub-id-type="pmid">38104648</pub-id></mixed-citation>
</ref>
<ref id="B44">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sourty</surname> <given-names>M.</given-names></name> <name><surname>Nasseef</surname> <given-names>M. T.</given-names></name> <name><surname>Champagnol-Di Liberti</surname> <given-names>C.</given-names></name> <name><surname>Mondino</surname> <given-names>M.</given-names></name> <name><surname>Noblet</surname> <given-names>V.</given-names></name> <name><surname>Parise</surname> <given-names>E. M.</given-names></name> <etal/></person-group>. (<year>2024b</year>). <article-title>Manipulating &#x003B4;fosb in d1-type medium spiny neurons of the nucleus accumbens reshapes whole-brain functional connectivity</article-title>. <source>Biol. Psychiatry</source> <volume>95</volume>, <fpage>266</fpage>&#x02013;<lpage>274</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.biopsych.2023.07.013</pub-id><pub-id pub-id-type="pmid">37517704</pub-id></mixed-citation>
</ref>
<ref id="B45">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Szpirer</surname> <given-names>C.</given-names></name></person-group> (<year>2020</year>). <article-title>Rat models of human diseases and related phenotypes: a systematic inventory of the causative genes</article-title>. <source>J. Biomed. Sci</source>. <volume>27</volume>:<fpage>84</fpage>. doi: <pub-id pub-id-type="doi">10.1186/s12929-020-00673-8</pub-id><pub-id pub-id-type="pmid">32741357</pub-id></mixed-citation>
</ref>
<ref id="B46">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Ortiz</surname> <given-names>R.</given-names></name> <name><surname>Chang</surname> <given-names>A.</given-names></name> <name><surname>Nasseef</surname> <given-names>T.</given-names></name> <name><surname>Rubalcaba</surname> <given-names>N.</given-names></name> <name><surname>Munson</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2025</year>). <article-title>Following changes in brain structure and function with multimodal mri in a year-long prospective study on the development of type 2 diabetes</article-title>. <source>Front. Radiol</source>. <volume>5</volume>:<fpage>1510850</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fradi.2025.1510850</pub-id><pub-id pub-id-type="pmid">40018732</pub-id></mixed-citation>
</ref>
<ref id="B47">
<mixed-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>F.</given-names></name> <name><surname>Koltun</surname> <given-names>V.</given-names></name></person-group> (<year>2015</year>). <article-title>Multi-scale context aggregation by dilated convolutions</article-title>. <source>arXiv preprint arXiv:1511.07122</source>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.1511.07122</pub-id></mixed-citation>
</ref>
</ref-list>
<fn-group>
<fn fn-type="custom" custom-type="edited-by" id="fn0001">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2676810/overview">Erica E. Jung</ext-link>, University of Illinois Chicago, United States</p></fn>
<fn fn-type="custom" custom-type="reviewed-by" id="fn0002">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/105865/overview">Teppei Matsui</ext-link>, Doshisha University Graduate School of Brain Science, Japan</p>
<p><ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3062780/overview">Zongpai Zhang</ext-link>, Johns Hopkins University, United States</p></fn>
</fn-group>
</back>
</article>