<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Archiving and Interchange DTD v2.3 20070202//EN" "archivearticle.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="methods-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1369717</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Methods</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title><italic>SE</italic>(3) group convolutional neural networks and a study on group convolutions and equivariance for DWI segmentation</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Liu</surname> <given-names>Renfei</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2628074/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lauze</surname> <given-names>Fran&#x000E7;ois</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2532065/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Bekkers</surname> <given-names>Erik J.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Darkner</surname> <given-names>Sune</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1086994/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Erleben</surname> <given-names>Kenny</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Department of Computer Science, University of Copenhagen</institution>, <addr-line>Copenhagen</addr-line>, <country>Denmark</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Computer Science, University of Amsterdam</institution>, <addr-line>Amsterdam</addr-line>, <country>Netherlands</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Alessandro Bria, University of Cassino, Italy</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Massimo Salvi, Polytechnic University of Turin, Italy</p>
<p>Siquan Wang, Columbia University, United States</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Kenny Erleben <email>kenny&#x00040;di.ku.dk</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>02</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1369717</elocation-id>
<history>
<date date-type="received">
<day>12</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>15</day>
<month>01</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Liu, Lauze, Bekkers, Darkner and Erleben.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Liu, Lauze, Bekkers, Darkner and Erleben</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>We present an <italic>SE</italic>(3) Group Convolutional Neural Network along with a series of networks with different group actions for segmentation of Diffusion Weighted Imaging data. These networks gradually incorporate group actions that are natural for this type of data, in the form of convolutions that provide equivariant transformations of the data. This knowledge provides a potentially important inductive bias and may alleviate the need for data augmentation strategies. We study the effects of these actions on the performances of the networks by training and validating them using the diffusion data from the Human Connectome project. Unlike previous works that use Fourier-based convolutions, we implement direct convolutions, which are more lightweight. We show how incorporating more actions - using the <italic>SE</italic>(3) group actions - generally improves the performances of our segmentation while limiting the number of parameters that must be learned.</p></abstract>
<kwd-group>
<kwd>geometric deep learning</kwd>
<kwd>group action</kwd>
<kwd>homogeneous spaces GCNN</kwd>
<kwd>image segmentation</kwd>
<kwd>diffusion weighted imaging</kwd>
</kwd-group>
<counts>
<fig-count count="9"/>
<table-count count="14"/>
<equation-count count="12"/>
<ref-count count="49"/>
<page-count count="20"/>
<word-count count="14018"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Medicine and Public Health</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>In this work, we study the influence of group actions on data and how they may impact the architecture and performances of neural networks, especially convolutional neural networks (CNN). CNNs rely on assumed translational symmetries in data and have shown very robust performance in imaging tasks, especially medical imaging ones, and they are highly parameter-efficient due to their weight-sharing property. When data offer more structure than simply translation, this can be used to build generalized CNNs. This is especially the case for the task at hand&#x02014;classification and segmentation of Diffusion Weighted Imaging (DWI) data. These Group and Geometric CNNs (GCNN) have been studied intensively and applied in many situations in the few past years (Masci et al., <xref ref-type="bibr" rid="B31">2015</xref>; Cohen and Welling, <xref ref-type="bibr" rid="B16">2016a</xref>; Boscaini et al., <xref ref-type="bibr" rid="B7">2016</xref>; Bekkers et al., <xref ref-type="bibr" rid="B5">2018</xref>; Cohen et al., <xref ref-type="bibr" rid="B15">2020</xref> to cite a few).</p>
<p>DWI is a non-invasive image modality that provides local information about water diffusion in tissues by means of measuring spin displacement (Tuchs, <xref ref-type="bibr" rid="B43">2004</xref>). It provides three-dimensional diffusion information at each location <italic>x</italic> that can be encoded as a function <italic>f</italic><sub><italic>x</italic></sub> on the two-dimensional sphere <italic>S</italic><sup>2</sup>. A field of these functions, on a given domain, can be represented as a function <italic>f</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;. If a sample is rotated and translated, the acquired signal should reflect, up to the limitations of acquisition protocol, this transformation. The group in question is the group of 3D rigid motions, <italic>SE</italic>(3), and the space &#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> is a <italic>homogeneous space</italic> under the action of <italic>SE</italic>(3): A point in &#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> can be transformed in any other point by a rigid transformation. This notion of homogeneous space is at the heart of the extension of CNNs to GCNNs (Cohen et al., <xref ref-type="bibr" rid="B15">2020</xref>; Bekkers, <xref ref-type="bibr" rid="B6">2019</xref>).</p>
<p>Our task at hand is the classification/segmentation of diffusion data. The inductive bias provided by the knowledge of these transformations may prove important for our task, especially when the amount of annotated data is limited. The problem boils down to how to incorporate this knowledge. The most classical approach is to use data augmentation, reflecting the expected symmetries in the data, in the hope that the network will be able to learn it during the training phase, learning symmetry-aware kernels.</p>
<p>Incorporating, on the other hand, some information about the symmetries of the data in the model has been shown to boost the performances of these networks (Bekkers et al., <xref ref-type="bibr" rid="B5">2018</xref>). But how much of this information is needed for a given task? To provide an answer, for the DWI segmentation task, we propose several networks, which gradually incorporate these symmetries in their architecture and study their performances. In addition, instead of performing convolution on non-Euclidean data in a spectral fashion using Fourier-type transformations, we implement convolution in all our experiments in a direct way, as is usually done in the image analysis community. In other words, we use regular representations of groups to encode the group actions in the models, instead of irreducible representations. Our experiments, in some sense, perform a <italic>group action ablation study</italic>. We start with a &#x0201C;naive&#x0201D; CNN and then incorporate spherical symmetries, resulting in a <italic>SO</italic>(3)-GCNN, discarding the spatial aspect of the data. The spatial aspect is then added in the form of a standard CNN coupled with spherical symmetries, and then, we build a network where roto-translational transformations are used in almost all steps. This work demonstrates empirically the improvement in performance. The results are, however, not always clear-cut. Previous works associated with group convolutions have addressed the capabilities of their models in comparison with data augmentation but, to the best of our knowledge, have not touched the comparison between models tested under randomly transformed test sets. This is what our ablation study is providing. It not only shows the impact of embedding transformations in the model but also gives a systematic analysis on the comparison among different group actions and the corresponding elements of network architecture with respect to the interplay between rotations and translations - the physically justified roto-translation group and the simpler direct product of translations and rotations - imposed in the models, and their relation to data augmentation, both in the training and test set. In the study we provide, the GCNN built from 3D-translations on one hand and rotations, on the other hand, seems to perform better than a <italic>SE</italic>(3)-GCNN. However, the <italic>SE</italic>(3)-network generalizes better to unseen rotated data than the previous one. The reason may lie in the particular type of data used - our DWI scans come from the Human Connectome Project (HCP) (Van Essen et al., <xref ref-type="bibr" rid="B44">2013</xref>) are highly preprocessed, including a form of alignment &#x02013; and this may impact the results. Nevertheless, for every model we propose, we also experiment training them with data augmentation to compare with our equivariant networks. We show that the more equivariance we incorporate into the model, the better the model resists the inconsistency of distributions between training and testing data.</p>
<p>The contribution of this work is as follows.</p>
<list list-type="bullet">
<list-item><p>We extend the prior work (Liu et al., <xref ref-type="bibr" rid="B30">2022</xref>) with a detailed theoretic formulation of the proposed method. We discretize <italic>SO</italic>(3) using the icosahedral rotation group and use rotation-translation separable filters in our model to make it very lightweight while achieving highly robust performance.</p></list-item>
<list-item><p>We provide an ablation study of different group actions in different spaces and the combinations of these actions with additional experiments using data augmentation.</p></list-item>
<list-item><p>We provide a comparison to M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>) in the experiments, which, to our knowledge, is the only other existing work that does tissue classification from DWI data using SE(3) group convolutions. In addition, we further provide experiments using the non-NN method of Schnell et al. (<xref ref-type="bibr" rid="B35">2009</xref>), which relies on rotationally invariant spherical harmonic (SH) features extracted from individual DWI voxels (squared-norms at given SH-orders), with classification performed by support vector machines (SVM). The spatial information is, however, discarded.</p></list-item>
</list>
<p>In the rest of this paper, we review related work, both around CNN and DWI classification problem. Then, we introduce the theoretical setup of GCNN and build several networks. Thereafter, we study and discuss their performances. Our implementation and experiments are publicly available at <ext-link ext-link-type="uri" xlink:href="https://github.com/rliu-p/se3gcnn">https://github.com/rliu-p/se3gcnn</ext-link>.</p>
</sec>
<sec id="s2">
<title>2 Related work</title>
<p>Deep Learning (DL) for non-flat data, or using more complex group actions than just translations, is currently getting more attention from the research field. When it comes to non-flat data, such as the point-wise spherical signals in DWI, particularly relevant related works are the following. A non-rotationally invariant modification was proposed by Boscaini et al. (<xref ref-type="bibr" rid="B7">2016</xref>). Schnell et al. (<xref ref-type="bibr" rid="B35">2009</xref>) developed an Support Vector Machine (SVM) using rotation-invariant features extracted from Spherical Harmonic decomposition of the HARDI signals, while Skibbe and Reisert (<xref ref-type="bibr" rid="B40">2017</xref>) introduced a toolkit for 3D image processing based on Spherical Tensor Algebra (STA), which is particularly well-suited for tasks requiring rotational invariance, such as image enhancement, reconstruction, and feature detection.</p>
<p>The above provide methods for DL-based processing of data on arbitrary manifolds. When the manifold, however, is a homogeneous space, i.e., there is a group action by which any two points on the manifolds can be reached, theory simplifies via a natural generalization of classical convolutions in group convolution neural networks (GCNNs), as was presented in Cohen et al. (<xref ref-type="bibr" rid="B14">2018</xref>); Bekkers et al. (<xref ref-type="bibr" rid="B5">2018</xref>); and Kondor and Trivedi (<xref ref-type="bibr" rid="B27">2018</xref>). GCNNs guarantee global equivariance. However, global equivariance can be complicated and elusive when the underlying geometry is non-trivial, which was discussed in Cohen et al. (<xref ref-type="bibr" rid="B17">2019</xref>). An elementary construction on a general manifold is proposed by Schonsheck et al. (<xref ref-type="bibr" rid="B36">2018</xref>) via a fixed choice of geodesic paths used to transport filters between points on the manifold, ignoring the effects of path dependency, i.e., holonomy when paths are geodesics. The removal of this path dependency can be obtained by summarizing local responses over local orientations, which is what was done by Masci et al. (<xref ref-type="bibr" rid="B31">2015</xref>). To explicitly deal with holonomy, Sommer and Bronstein (<xref ref-type="bibr" rid="B42">2020</xref>) proposed a theoretical breakthrough using convolution construction on manifolds based on stochastic processes via the frame bundle.</p>
<p>On the other hand, Cohen et al. (<xref ref-type="bibr" rid="B14">2018</xref>) lifted spherical functions to the 3D-rotation group <italic>SO</italic>(3) and used a generalization of Fourier transform on it to perform convolution. Elaldi et al. (<xref ref-type="bibr" rid="B20">2021</xref>) proposed an equivariant spherical deconvolution method to learn the orientation distribution function (ODF). Bouza et al. (<xref ref-type="bibr" rid="B8">2021</xref>) generalized convolution to manifold-valued convolutions using Volterra Series, preserving its equivariance. With the generalization of convolution to more complex group actions than translation, several authors (Gens and Domingos, <xref ref-type="bibr" rid="B21">2014</xref>; Cohen and Welling, <xref ref-type="bibr" rid="B16">2016a</xref>; Weiler et al., <xref ref-type="bibr" rid="B47">2018b</xref>,<xref ref-type="bibr" rid="B46">a</xref>; Worrall et al., <xref ref-type="bibr" rid="B48">2017</xref>; Kondor and Trivedi, <xref ref-type="bibr" rid="B27">2018</xref>; Bekkers et al., <xref ref-type="bibr" rid="B5">2018</xref>; Andrearczyk et al., <xref ref-type="bibr" rid="B1">2020</xref>; Chakraborty et al., <xref ref-type="bibr" rid="B10">2018a</xref>,<xref ref-type="bibr" rid="B11">b</xref>, <xref ref-type="bibr" rid="B12">2020</xref>; Graham et al., <xref ref-type="bibr" rid="B24">2020</xref>) explored the group convolution path for Lie groups and the homogeneous spaces of these groups. Knigge et al. (<xref ref-type="bibr" rid="B26">2022</xref>) proposed a separable convolution setup on Lie groups. The relation between group actions, principal bundles and related vector bundles, and convolutional architectures is currently explored (Cohen et al., <xref ref-type="bibr" rid="B17">2019</xref>, <xref ref-type="bibr" rid="B15">2020</xref>; Aronsson, <xref ref-type="bibr" rid="B2">2022</xref>). The latter elucidates important relations between differential geometry of bundles and Reproducible Kernel Hilbert Spaces. Links between partial differential equations, symmetries, and GCNN are studied in Smets et al. (<xref ref-type="bibr" rid="B41">2021</xref>). A unifying framework for equivariant DL on manifolds, connecting both the bundle and homogeneous space viewpoint, is given in Weiler et al. (<xref ref-type="bibr" rid="B45">2021</xref>) through a notion of coordinate indepencent convolutions.</p>
<p>Most CNNs approach for the processing of DWI signals discards its specific structure. For instance, Golkov et al. (<xref ref-type="bibr" rid="B23">2016</xref>) built multi-layer perceptrons in <italic>q</italic>-space for kurtosis and NODDI mappings. However, the importance of spherical equivariant or invariant structure has been acknowledged for some years now. The importance of the extraction of rotationally invariant features beyond Fractional Anisotropy (Basser et al., <xref ref-type="bibr" rid="B4">1994</xref>) has been recognized in series of DWI works. For instance, Caruyer and Verma (<xref ref-type="bibr" rid="B9">2015</xref>) developed invariant polynomials of spherical harmonic (SH) expansion coefficients and discussed their application in population studies. Schwab et al. (<xref ref-type="bibr" rid="B37">2013</xref>) proposed a related construction using eigenvalue decomposition of SH operators. Novikov et al. (<xref ref-type="bibr" rid="B33">2018</xref>) and Zucchelli et al. (<xref ref-type="bibr" rid="B49">2020</xref>) argued their usefulness for understanding microstructures in relation to DWI.</p>
<p>Chakraborty et al. (<xref ref-type="bibr" rid="B10">2018a</xref>) proposed a rotation equivariant construction inspired by Cohen et al. (<xref ref-type="bibr" rid="B14">2018</xref>) for disease classification. The same authors (Banerjee et al., <xref ref-type="bibr" rid="B3">2019</xref>) used a <italic>S</italic><sup>2</sup>&#x000D7;&#x0211D;<sup>&#x0002B;</sup> CNN using SHORE function representation for classification in Parkinson&#x00027;s Disease. Sedlar et al. (<xref ref-type="bibr" rid="B39">2020</xref>) used a spherical U-Net for f-ODF estimation. The same authors (Sedlar et al., <xref ref-type="bibr" rid="B38">2021</xref>) used a spherical CNN for microstructure parameter estimation, using spherical harmonics representations. M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>) proposed a sixth-D, 3D space and <italic>q</italic>-space NNs with roto-translation/rotation equivalence properties, targeted at DWI data. Poulenard et al. (<xref ref-type="bibr" rid="B34">2022</xref>) reviewed several implementations of <italic>SE</italic>(3) neural networks and showcased a comparison among these networks. In their work, steerable CNNs generalize better than group CNNs while dealing with inconsistent distributions between training and testing data for 3D images.</p>
<p>While most equivariant methods use spectral representation of groups, we propose an SE(3) network for DWI data that uses <italic>regular representation</italic> of groups such that the whole model is light-weight, and the implementation for convolution is not only direct but also separable, improving efficiency. A similar idea was used in Chen et al. (<xref ref-type="bibr" rid="B13">2021</xref>), for 3D point cloud feature extraction, with, however, important architectural differences due to the nature of input data. Both our method and Chen et al. (<xref ref-type="bibr" rid="B13">2021</xref>) implemented regular representations of groups in a separable fashion; however, their separable kernels are only over the spatial and rotation interactions while we additionally split the rotation interactions over 2 axes, making use of the factorization of the icosahedron group into 12 &#x000D7; 5 rotations. We do this to further boost efficiency. Furthermore, Chen et al. (<xref ref-type="bibr" rid="B13">2021</xref>) include an attention mechanism in the interaction layers, while instead, we use non-linearities between the separate interaction steps. As our operations are strictly local, including attention mechanism would introduce unnecessary computational overhead, whereas in Chen et al. (<xref ref-type="bibr" rid="B13">2021</xref>) the attention mechanism could be critical as a selection mechanism among the global interactions between many points within the point cloud. In addition, we compared our method to M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>) which uses steerable filter bases (spectral representation of groups) for the <italic>SE</italic>(3) group. In our experiments, in comparison with M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>), we found out, however, that our direct convolution implementation of <italic>SE</italic>(3) GCNN does not perform inferior to its steerable alternative, and our method is a lot more light-weight.</p>
</sec>
<sec id="s3">
<title>3 Method</title>
<p>The networks we present will be built from the principle of expanding CNNs to groups and their homogeneous spaces, on which they act by extending convolution operations to functions on groups and their homogeneous spaces. For the rotation group <italic>SO</italic>(3) and the sphere <italic>S</italic><sup>2</sup> as <italic>SO</italic>(3)-homogeneous space, the common path for implementing convolutions/correlations is to use irreducible representations (Cohen and Welling, <xref ref-type="bibr" rid="B18">2016b</xref>). This approach can be computationally very intensive, unless one restricts to very low-order irreducible representations, with a resolution trade-off worse than the approximation of <italic>SO</italic>(3) by the icosahedral rotation group. So we do not follow that path here.</p>
<p>In the next section, we provide the theoretical background for extending convolutions to functions on groups. For the reader&#x00027;s convenience, standard concepts from group theory and group actions that are used to build our new convolution layers are presented in <xref ref-type="supplementary-material" rid="SM1">Appendix 1</xref>.</p>
<sec>
<title>3.1 Generalized convolution operations</title>
<p>Classical CNNs use the standard convolution operation on &#x0211D;<sup><italic>n</italic></sup>: for <italic>h</italic>, &#x003BA;:&#x0211D;<sup><italic>n</italic></sup> &#x02192; &#x0211D;, where <italic>h</italic> is the signal and &#x003BA; is the kernel, the operation is defined as</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>h</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mstyle><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mstyle mathvariant="bold"><mml:mi>y</mml:mi></mml:mstyle><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:msub><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here &#x0211D;<sup><italic>n</italic></sup> is the underlying space of the function, <italic>n</italic> &#x0003D; 2 for 2D images, and <italic>n</italic> &#x0003D; 3 for 3D volumetric images. This operation can be extended to vector-valued functions (i.e., functions from &#x0211D;<sup><italic>n</italic></sup> &#x02192; &#x0211D;<sup><italic>m</italic></sup>, they have <italic>m</italic> <italic>channels</italic>) and multiple kernels, and this is of course at the heart of the definition of a convolutional layer in a CNN.</p>
<p>Rewrite <xref ref-type="disp-formula" rid="E1">Equation 1</xref> as</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>h</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mstyle mathvariant="double-struck"><mml:mi>T</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:msub></mml:mstyle><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mi>&#x003BA;</mml:mi><mml:mo>:</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>&#x021A6;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p><italic>L</italic><sub><italic>y</italic></sub>&#x003BA; <italic>translates</italic> the kernel &#x003BA; by vector <italic>y</italic>. This is the <italic>left regular representation</italic> of &#x1D54B;<sup><italic>n</italic></sup> on the space of kernels (see <xref ref-type="supplementary-material" rid="SM1">Appendix 1.2.5</xref>). Using the regular representation, one gets that the standard convolution (<xref ref-type="disp-formula" rid="E1">Equation 1</xref>) is <italic>translation-equivariant</italic>, a property generally acknowledged as the main source of success for CNNs:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mi>h</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>h</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mo>-</mml:mo><mml:mstyle mathvariant="bold"><mml:mtext>z</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In <xref ref-type="supplementary-material" rid="SM1">Appendix 1.2.5</xref>, a general definition for the regular representation is given for a Lie group <italic>G</italic> acting on a homogeneous space &#x02133;. It is defined by <inline-formula><mml:math id="M4"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mi>f</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. This is in particular the case when &#x02133; is a principal homogeneous space of <italic>G</italic>, and especially when &#x02133; &#x0003D; <italic>G</italic>. This leads to a generalization of convolutions for functions defined on a group <italic>G</italic>: if <italic>h</italic>, &#x003BA;:<italic>G</italic> &#x02192; &#x0211D;, <italic>h</italic>&#x0002A;<sub><italic>G</italic></sub>&#x003BA;, or simply <italic>h</italic>&#x0002A;&#x003BA;, if there is no ambiguity, is defined by</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>h</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mstyle><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi></mml:mrow></mml:msub><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>u</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mstyle><mml:mi>h</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>u</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>u</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here, <italic>du</italic> refers to a Haar measure in <italic>G</italic> (Diestel and Spalsbury, <xref ref-type="bibr" rid="B19">2014</xref>). This operation is equivariant to transformations in the group with respect to the regular representation essentially exactly as in <xref ref-type="disp-formula" rid="E3">Equation 3</xref>:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x02200;</mml:mo><mml:mi>v</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>G</mml:mi><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mi>h</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>h</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>h</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This operation is also equivariant for the left-representation of G.</p>
<p>We deal rarely directly with functions whose domain is a non-trivial group, such as <italic>SE</italic>(3), or data &#x0201C;indexed&#x0201D; by a non-trivial group. The domain is instead a homogeneous space of the group of interest, such as &#x0211D;<sup>3</sup> or the sphere <italic>S</italic><sup>2</sup> for the groups in this work. In that situation, kernel convolution generalizes to a <italic>lifting</italic> operation that produces a new function, this time defined on the group. If <italic>f</italic>:&#x02133; &#x02192; &#x0211D; and <italic>k</italic>:&#x02133; &#x02192; &#x0211D; are the function and the kernel, respectively, define <italic>f</italic>&#x0002A;<sub>lifting</sub><italic>k</italic>, or just <italic>f</italic>&#x0002A;<italic>k</italic>, if there is no ambiguity, by</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mi>&#x02133;</mml:mi></mml:mstyle></mml:mrow></mml:msub></mml:mstyle><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>m</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mi>&#x02133;</mml:mi></mml:mstyle></mml:mrow></mml:msub></mml:mstyle><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>m</mml:mi><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>and this convolution operation is equivariant with respect to actions in <italic>G</italic>:</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mo>&#x02200;</mml:mo><mml:mi>v</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>G</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mi>f</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002A;</mml:mo><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>v</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>f</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>k</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>A bit of caution here, as the first regular representation acts on a function <italic>f</italic>:&#x02133; &#x02192; &#x0211D; while the second acts on the function <italic>f</italic>:<italic>G</italic> &#x02192; &#x0211D;. Once the function is lifted onto this group <italic>G</italic>, <italic>group convolutions</italic> on <italic>G</italic> can be performed on the lifted signals as in <xref ref-type="disp-formula" rid="E4">Equation 4</xref>,</p>
<p>The group convolutions and lifting can be stacked in layers like a standard CNN, and this stacking preserves equivariance, producing equivariant layers. Features at the last group convolution layer can be <italic>projected</italic> back onto the original space of the function by summarizing feature responses over the group. It is similar to max-pooling-like operations in a standard CNN. This type of operation will provide invariance.</p>
<p>Therefore, a roadmap for <italic>group convolutions</italic> can be summarized as follows:</p>
<list list-type="bullet">
<list-item><p>Lifting the function signals to the desired group.</p></list-item>
<list-item><p>Group convolutions on the lifted signals.</p></list-item>
<list-item><p>Projecting the signals back onto the original space.</p></list-item>
</list>
<p>We formulate these operations in the following sections.</p>
<sec>
<title>3.1.1 Lifting layer</title>
<p>A function <inline-formula><mml:math id="M9"><mml:mi>f</mml:mi><mml:mo>:</mml:mo><mml:mstyle mathvariant="bold"><mml:mi>&#x02133;</mml:mi></mml:mstyle><mml:mo>&#x02192;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> can be <italic>lifted</italic> to the group <italic>G</italic> via a kernel <inline-formula><mml:math id="M10"><mml:mi>&#x003BA;</mml:mi><mml:mo>:</mml:mo><mml:mstyle mathvariant="bold"><mml:mi>&#x02133;</mml:mi></mml:mstyle><mml:mo>&#x02192;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> by</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mstyle mathvariant="bold"><mml:mi>&#x02133;</mml:mi></mml:mstyle></mml:mrow></mml:msub></mml:mstyle><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msubsup></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This is a direct extension of <xref ref-type="disp-formula" rid="E6">Equation 6</xref> to vector-valued functions <italic>f</italic>. <italic>N</italic><sub>0</sub> is the number of input channels, and <italic>N</italic><sub>1</sub> the number of output channels. In practice, in this work, the input function is scalar-valued, i.e., <italic>N</italic><sub>0</sub> &#x0003D; 1.</p>
</sec>
<sec>
<title>3.1.2 Group convolution layer</title>
<p>A feature function <inline-formula><mml:math id="M12"><mml:mi>F</mml:mi><mml:mo>:</mml:mo><mml:mi>G</mml:mi><mml:mo>&#x02192;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> is transformed by a convolution kernel <inline-formula><mml:math id="M13"><mml:mi>K</mml:mi><mml:mo>:</mml:mo><mml:mi>G</mml:mi><mml:mo>&#x02192;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:math></inline-formula> by</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M14"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mo>&#x0002A;</mml:mo><mml:mi>K</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mstyle><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:mi>h</mml:mi></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msubsup><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Here <italic>N</italic><sub><italic>l</italic></sub> is the number of channels from the output of the last layer (equivalent to the number of input channels for the current layer), and <italic>N</italic><sub><italic>l</italic>&#x0002B;1</sub> is the number of output channels for the current layer.</p>
</sec>
<sec>
<title>3.1.3 Projection layer</title>
<p>If needed, feature map <italic>F</italic>:<italic>G</italic> &#x02192; &#x0211D;<sup><italic>n</italic></sup> can be projected to a function <italic>f</italic>:&#x02133; &#x02192; &#x0211D;<sup><italic>n</italic></sup> by summarizing on the fibers (see <xref ref-type="supplementary-material" rid="SM1">appendix 1.2.3</xref>.)</p>
<disp-formula id="E10"><label>(10)</label><mml:math id="M15"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo class="qopname">max</mml:mo></mml:mrow><mml:mrow><mml:mi>h</mml:mi><mml:mo>&#x02208;</mml:mo><mml:msub><mml:mrow><mml:mi>G</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mrow></mml:munder></mml:mstyle><mml:mi>F</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi><mml:mi>h</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mtext>&#x02003;</mml:mtext><mml:mtext class="textrm" mathvariant="normal">for any&#x000A0;</mml:mtext><mml:mi>g</mml:mi><mml:mtext class="textrm" mathvariant="normal">&#x000A0;with&#x000A0;</mml:mtext><mml:mi>g</mml:mi><mml:mo>.</mml:mo><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>m</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the max is computed component-wise. This operation is equivariant: <inline-formula><mml:math id="M16"><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mi>F</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mover accent="false" class="mml-overline"><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mo accent="true">&#x000AF;</mml:mo></mml:mover></mml:math></inline-formula>.</p>
</sec>
<sec>
<title>3.1.4 Activation functions and separable kernels</title>
<p>A point-wise activation function &#x003B1;, such as ReLU, is trivially equivariant <italic>L</italic><sub><italic>g</italic></sub>(&#x003B1;<italic>f</italic>) &#x0003D; &#x003B1;(<italic>L</italic><sub><italic>g</italic></sub><italic>f</italic>). On manifolds with an underlying product structure, &#x02133; &#x0003D; &#x02133;<sub>1</sub>&#x000D7;&#x02133;<sub>2</sub> - this includes homogeneous spaces and groups - one can choose separable kernels &#x003BA; &#x0003D; &#x003BA;<sub>&#x02133;<sub>1</sub></sub>&#x02297;&#x003BA;<sub>&#x02133;<sub>2</sub></sub>, and activation functions can be intertwined in between <xref ref-type="disp-formula" rid="E8">Equations 8</xref>, <xref ref-type="disp-formula" rid="E9">9</xref>. For instance, lifting <xref ref-type="disp-formula" rid="E8">Equation 8</xref> can be replaced by</p>
<disp-formula id="E11"><label>(11)</label><mml:math id="M17"><mml:mtable class="eqnarray" columnalign="right"><mml:mtr><mml:mtd><mml:mi>f</mml:mi><mml:msup><mml:mrow><mml:mo>&#x0002A;</mml:mo></mml:mrow><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow></mml:msup><mml:mi>&#x003BA;</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mi>&#x02133;</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mstyle></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>&#x003B1;</mml:mi><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mstyle displaystyle="true"><mml:msub><mml:mrow><mml:mo>&#x0222B;</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mstyle mathvariant="bold"><mml:mi>&#x02133;</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:mstyle><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003BA;</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>d</mml:mi><mml:msub><mml:mrow><mml:mi>m</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x0002A;<sup>&#x003B1;</sup>&#x003BA; is a shortcut notation for the intertwining of the kernel and activation function. It is easily seen that it preserves equivariance. Having separable kernels increases the efficiency of the model since it increases weight sharing. For example, instead of having kernels defined in &#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup>, we have kernels defined in &#x0211D;<sup>3</sup> and in <italic>S</italic><sup>2</sup>. In this way, all voxels in &#x0211D;<sup>3</sup> share the same spherical kernels. This is used in this work.</p>
<p>The spaces used in this work are &#x0211D;<sup>3</sup>, the sphere <italic>S</italic><sup>2</sup>, and the product space &#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup>. The groups that we consider are the group of translations of &#x0211D;<sup>3</sup>, &#x1D54B;<sup>3</sup> &#x02243; &#x0211D;<sup>3</sup>, the group <italic>SO</italic>(3) or 3D rotations, the direct product &#x1D4A2; &#x0003D; &#x1D54B;<sup>3</sup> &#x000D7; <italic>SO</italic>(3), and the special Euclidean group <italic>SE</italic>(3) &#x0003D; <italic>SO</italic>(3)&#x022C9;&#x1D54B;<sup>3</sup>. Note that though &#x1D4A2; and <italic>SE</italic>(3) are isomorphic as manifolds, they are not as groups: in &#x1D4A2;, <inline-formula><mml:math id="M19"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mi>R</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mi>S</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover><mml:mo>&#x0002B;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mi>R</mml:mi><mml:mi>S</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> while in <italic>SE</italic>(3), <inline-formula><mml:math id="M20"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mo>,</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>.</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mo>,</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover><mml:mo>&#x0002B;</mml:mo><mml:mi>R</mml:mi><mml:mover accent="true"><mml:mrow><mml:mi>s</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula>. This is also reflected in their respective actions in <xref ref-type="table" rid="T1">Table 1</xref>, which shows the different combinations of spaces and groups. We refer the readers to Gerken et al. (<xref ref-type="bibr" rid="B22">2023</xref>) for more detailed theoretical foundation.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Groups and homogeneous spaces in this work.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0001.tif"/></th>
<th valign="top" align="center"><bold>&#x0211D;<sup>3</sup>, <italic>x</italic></bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M21"><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:math></inline-formula></bold></th>
<th valign="top" align="center"><bold><inline-formula><mml:math id="M22"><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msup><mml:mo>&#x000D7;</mml:mo><mml:msup><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M23"><mml:msup><mml:mrow><mml:mstyle mathvariant="double-struck"><mml:mi>T</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msup><mml:mo>,</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M24"><mml:mi>x</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:math></inline-formula></td>
<td/>
<td/>
</tr>
<tr>
<td valign="top" align="left"><italic>SO</italic>(3), <italic>R</italic></td>
<td/>
<td valign="top" align="center"><inline-formula><mml:math id="M25"><mml:mi>R</mml:mi><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:math></inline-formula></td>
<td/>
</tr>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M26"><mml:msup><mml:mrow><mml:mstyle mathvariant="double-struck"><mml:mi>T</mml:mi></mml:mstyle></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msup><mml:mo>&#x000D7;</mml:mo><mml:mi>S</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mi>R</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M27"><mml:mi>x</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M28"><mml:mi>R</mml:mi><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M29"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mi>R</mml:mi><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr>
<tr>
<td valign="top" align="left"><inline-formula><mml:math id="M30"><mml:mi>S</mml:mi><mml:mi>E</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mo>,</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M31"><mml:mi>R</mml:mi><mml:mi>x</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M32"><mml:mi>R</mml:mi><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:math></inline-formula></td>
<td valign="top" align="center"><inline-formula><mml:math id="M33"><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mi>x</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover><mml:mo>,</mml:mo><mml:mi>R</mml:mi><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula></td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>For each group and each homogeneous space, typical elements are provided, as well as the action of the group element on the space element. Entries left empty are not used or fail to be homogeneous spaces for standard group actions on them.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec>
<title>3.2 Discretization of spherical signals</title>
<p>The way spherical signals are numerically handled have major implications for our networks. A DWI signal is treated as a discretization of a signal <italic>f</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;. DWIs are acquired, for each voxel, at <italic>N</italic> fixed directions <italic>p</italic><sub>1</sub>, &#x02026;, <italic>p</italic><sub><italic>N</italic></sub> on <italic>S</italic><sup>2</sup> (here <italic>N</italic> &#x0003D; 90). These are represented in two different ways.</p>
<list list-type="bullet">
<list-item><p>Type 1. Ignoring the spherical structure, at each voxel <italic>x</italic>, we get a measurement vector</p>
<p><inline-formula><mml:math id="M34"><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02208;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x0211D;</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula>. Thus an image is a mapping <italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup>.</p>
</list-item>
<list-item><p>Type 2. A signal at voxel <italic>x</italic> is interpolated as a proper spherical function <inline-formula><mml:math id="M35"><mml:mi>f</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi><mml:mo>,</mml:mo><mml:mover accent="true"><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mo>&#x02192;</mml:mo></mml:mover></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>W</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>v</mml:mi><mml:mo>;</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x02026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>v</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:math></inline-formula> where <italic>W</italic> is a Watson kernel (Jupp and Mardia, <xref ref-type="bibr" rid="B25">1989</xref>). An image from this type is a mapping <italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;.</p></list-item>
</list>
</sec>
<sec>
<title>3.3 Direct convolution and discretization of groups</title>
<p>Unlike existing methods that use generalized Fourier-type transforms to perform convolution on spheres (Cohen et al., <xref ref-type="bibr" rid="B14">2018</xref>; Gens and Domingos, <xref ref-type="bibr" rid="B21">2014</xref>; Cohen and Welling, <xref ref-type="bibr" rid="B16">2016a</xref>; Weiler et al., <xref ref-type="bibr" rid="B46">2018a</xref>; Worrall et al., <xref ref-type="bibr" rid="B48">2017</xref>; Kondor and Trivedi, <xref ref-type="bibr" rid="B27">2018</xref>; Bekkers et al., <xref ref-type="bibr" rid="B5">2018</xref>; Andrearczyk et al., <xref ref-type="bibr" rid="B1">2020</xref>; Chakraborty et al., <xref ref-type="bibr" rid="B10">2018a</xref>,<xref ref-type="bibr" rid="B11">b</xref>, <xref ref-type="bibr" rid="B12">2020</xref>), we implement the convolution for spheres directly as in classical 2D CNNs in the image analysis field. We first discretize the sphere <italic>S</italic><sup>2</sup> using an icosahedron. To lift the function from the sphere to the <italic>SO</italic>(3) group, we define a star-shaped kernel <italic>k</italic>:<italic>S</italic><sup>2</sup>&#x021A6;&#x0211D; with a limited support. The kernel then moves around the discretized sphere and convolves with signals at each vertex of the icosahedron. It rotates five times at each icosahedral vertex according to the fives edges each vertex has, and collects convolutional responses from all five rotations. In this way, the spherical function is lifted to <italic>SO</italic>(3), which is discretized by <italic>I</italic><sub><italic>SO</italic>(3)</sub>&#x02014;the 60 rotational symmetries of an icosahedron. This corresponds to <xref ref-type="disp-formula" rid="E8">Equation 8</xref> and is shown in <xref ref-type="fig" rid="F1">Figure 1A</xref>. For the <italic>SO</italic>(3) group convolution layer, the kernel is defined on <italic>SO</italic>(3), which is represented by the icosahedral symmetries. Here, we specially design the kernel in the way that the support of it covers exactly a fiber. Therefore, we rotate (permute) the kernel at each fiber, convolve the rotated kernels with the fiber, and move the kernel to the next fiber. This is how <xref ref-type="disp-formula" rid="E9">Equation 9</xref> is implemented, and more details can be found in <xref ref-type="fig" rid="F1">Figure 1B</xref>. With the impact of discretization of the groups and the interpolation of signals, we lose the benefits of learning from raw data. However, the experiments show that the 60 icosahedral symmetries can approximate the <italic>SO</italic>(3) group well enough such that the models can deal with rotational variations in the data that are different from the rotations used in the discretization.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Three group convolution operators used in this paper. <bold>(A)</bold> shows the spherical part of the separable lifting convolution. The star-shaped kernel translates (in this case translation is equivalent to rotation) to the 12 icosahedron vertices like a spider crawling on a sphere. At each vertex location, the kernel rotates five times aligned with the edges of the icosahedron and gets five responses from all the orientations. Therefore, at each vertex, the output is a fiber consisting of five elements. There are in total 60 responses from all 12 vertices, and thus 60 rotation matrices to translate the kernel, assembling a discretization of <italic>SO</italic>(3) - <italic>I</italic><sub><italic>SO</italic>(3)</sub>. <bold>(B)</bold> shows the spherical part of the separable group convolution. The kernel is then defined at each fiber and is rotated (permuted) again for five times to get the responses of different orientations, as in the lifting convolution. <bold>(C)</bold> shows the spatial part of the separable convolution (the spatial convolution is the same in the lifting and group convolution; thus, we only show one). The spatial kernel is a 3D grid. The grid is rotated to convolve with all 60 spherical responses. The kernel is rotated 60 times, using the same icosahedral symmetry rotations as those on which the input is sampled.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0001.tif"/>
</fig>
</sec>
<sec>
<title>3.4 Generic networks used in this work</title>
<p>We present four constructions in which gradual levels of complexity in group actions are introduced. This can be seen as a group action ablation study. The precise description of each network will be provided in Section 4.</p>
<sec>
<title>3.4.1 Group of translations &#x1D54B;<sup>3</sup></title>
<p>The <italic>S</italic><sup>2</sup>-structure of the signal is ignored, using the Type 1 discretization. The group being &#x1D54B;<sup>3</sup>, just another name for &#x0211D;<sup>3</sup>, we just obtain a standard CNN, ignoring rotational information. An illustration can be found in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Illustration of the classical CNN. In the grids shown above, which assembles the dimensions of feature maps in the later experiments. Each voxel in the <italic>ith</italic> layer contains <italic>C</italic><sub><italic>i</italic></sub> values, indicating the numbers of channels. <italic>C</italic><sub>1</sub> here is the number of signal values each voxel from the original scan, thus 90. Due to striding, the grid shrinks to 1 voxel after 3 convolutional layers and then is fed into a fully connected layer for classification.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0002.tif"/>
</fig>
</sec>
<sec>
<title>3.4.2 <italic>SO</italic>(3)</title>
<p>This time the spatial structure is ignored, and each voxel provides a spherical data point. Type 2 discretization is used. The GCNN takes as input a spherical function and will classify it by performing <italic>SO</italic>(3)-lifting, <italic>SO</italic>(3)-convolutions and summarization. The convolved function on <italic>SO</italic>(3) is then projected back to <italic>S</italic><sup>2</sup> by this summarization. It is illustrated in <xref ref-type="fig" rid="F1">Figures 1A</xref>, <xref ref-type="fig" rid="F1">B</xref>. This model is a fully equivariant implementation of <italic>SO</italic>(3) group convolution followed by the work in Liu et al. (<xref ref-type="bibr" rid="B29">2021</xref>), which does not hold global equivariance.</p>
</sec>
<sec>
<title>3.4.3 &#x1D54B;<sup>3</sup> &#x000D7; <italic>SO</italic>(3)</title>
<p>Spatial and spherical structures are decoupled. This implies a standard spatial CNN dealing with only voxel translations, and a <italic>SO</italic>(3)-GCNN part for the directional signal. Type 2 discretization is used for spherical signals. The decoupled &#x0211D;<sup>3</sup>-layer and <italic>S</italic><sup>2</sup>-layer are with group actions &#x1D54B;<sup>3</sup> and <italic>SO</italic>(3), respectively. The illustration for the <italic>S</italic><sup>2</sup>-layer can be found in <xref ref-type="fig" rid="F1">Figures 1A</xref>, <xref ref-type="fig" rid="F1">B</xref>, and the illustration for the &#x0211D;<sup>3</sup>-layer can be regarded as only one Conv3D operation in <xref ref-type="fig" rid="F1">Figure 1C</xref> without the rotations. Note that since the spatial convolution does not incorporate rotational equivariance, it does not reflect equivariance of the DWI measurements. I.e., one can expect that when the brain rotates, the spatial patterns rotate, as well as their spherical diffusion signals. This model takes rotation into account in the spherical part of the signal but not the spatial part. The projection at the end collapses the function in the group back to &#x0211D;<sup>3</sup> by summarizing&#x02014;in this case, maximizing&#x02014;over <italic>SO</italic>(3), and the resulting feature map is fed into a fully connected layer to perform the classification task.</p>
</sec>
<sec>
<title>3.4.4 <italic>SE</italic>(3)</title>
<p>Type 2 discretization is used, and the network uses the full interplay between spatial roto-translations and corresponding rotations of the spherical signal and is thus fully equivariant to <italic>SE</italic>(3) transformations on the DWI data. <xref ref-type="fig" rid="F1">Figures 1A</xref>, <xref ref-type="fig" rid="F1">B</xref> shows the kernels of the <italic>S</italic><sup>2</sup>-layer. When the kernel moves from one vertex to another, it follows a specific rotation that maps the one-ring neighborhood of the source vertex to the one-ring neighborhood of the target vertex. At each vertex, the kernel has an <italic>SO</italic>(2) symmetry group structure discretized by 5 rotations. <xref ref-type="fig" rid="F1">Figure 1C</xref> shows the kernel for the &#x0211D;<sup>3</sup>-layer. It is rotated with the same rotation matrices that moved the <italic>S</italic><sup>2</sup>-kernel as in <xref ref-type="fig" rid="F1">Figures 1A</xref>, <xref ref-type="fig" rid="F1">B</xref>. Since the spatial kernels are cube-shaped grids, interpolation is required while rotating them. Here, we use linear interpolation, which can be easily implemented. To perform the segmentation task, the projection layer collapses the function on <italic>SE</italic>(3) back to &#x0211D;<sup>3</sup> by summarizing&#x02014;again, maximizing&#x02014;over <italic>SO</italic>(3).</p>
</sec>
</sec>
</sec>
<sec id="s4">
<title>4 Experiments and results</title>
<p>In this section, we first list all the detailed network setups, after which we present the results of the experiments. We evaluate our method on the DWI brain dataset from the human connectome project (HCP) (Van Essen et al., <xref ref-type="bibr" rid="B44">2013</xref>). We classify the human brains into four regions - cerebrospinal fluid (CSF), subcortical, white matter (WM), and gray matter (GM). An illustration of the task can be found in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Left to right: original diffusion data, the ground truth segmentation, and the processed ground-truth that we are going to learn from. The label colors for CSF, subcortical, white matter, and gray matter are red, blue, white, and gray, respectively. The figures only illustrate the data, and they are not necessarily from the same slice of the same scan.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0003.tif"/>
</fig>
<p>We use the preprocessed DWI data (Van Essen et al., <xref ref-type="bibr" rid="B44">2013</xref>) and normalize each DWI scan for the <italic>b</italic>-1000 images with the voxel-wise average of the <italic>b</italic><sub>0</sub>. We use the brain masks provided in the dataset to obtain the voxels of interest, while background is ignored. The labels provided with the T1-image are transformed to the DWI using nearest neighbor interpolation (<xref ref-type="fig" rid="F3">Figure 3</xref>). The resolution of the DWI images is 145 &#x000D7; 174 &#x000D7; 145, and the resolution of the T1-images is 260 &#x000D7; 311 &#x000D7; 260. Focal Loss (Lin et al., <xref ref-type="bibr" rid="B28">2018</xref>) is used to counter the class imbalance of the four brain regions. For Focal Loss, all experiments use &#x003B3; &#x0003D; 2 and use &#x003B1; &#x0003D; (0.35, 0.35, 0.15, 0.15) for CSF, subcortical, WM, and GM, respectively. For the Watson Kernel, all experiments that used this interpolation (Type 2 discretization) have &#x003BA; &#x0003D; 10. Batch size for all experiments is 100, and the learning rate for all experiments is 0.001.</p>
<sec>
<title>4.1 Experimental setup</title>
<p>Since each DWI scan is highly resoluted, it is not feasible to use a whole image as input to the networks. Therefore, to reduce the computational burden, as inputting a full DWI volume is intractable, we use spatial windows of <italic>N</italic><sup>3</sup> voxels, with <italic>N</italic> &#x0003D; 1 for the <italic>SO</italic>(3)-action network and <italic>N</italic> &#x0003D; 7 for the rest. In addition, due to the effect of striding in spatial convolution, the 7<sup>3</sup> grid of voxels shrinks to 1<sup>3</sup> after 3 spatial convolutions. Therefore, a separable convolution layer (for both &#x1D54B;<sup>3</sup> &#x000D7; <italic>SO</italic>(3) and <italic>SE</italic>(3) actions) is equivalent to a single <italic>SO</italic>(3) convolution layer when the grid shrinks to 1<sup>3</sup> since the spatial convolution becomes trivial. <italic>S</italic><sup>2</sup> is discretized by a regular icosahedron. <italic>SO</italic>(3) is discretized as the icosahedral rotation group with 60 elements. Each vertex of the icosahedron is fixed by five rotations, isomorphic to the subgroup of <italic>SO</italic>(2) consisting of rotations of angle 2<italic>k&#x003C0;</italic>/5, <italic>k</italic> &#x0003D; 0&#x02026;4. This is, of course, the discretization used for <italic>SO</italic>(2).</p>
<p>To validate the proposed <italic>SE</italic>(3) network, we first provide an ablation study of our proposed four types of networks based on different group actions. Then, we compare it with M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>), which implements an <italic>SE</italic>(3)-GCNN using <italic>irreducible representations</italic>.</p>
<p>For the ablation study, based on the networks that were introduced above and in alignment with the networks presented in Liu et al. (<xref ref-type="bibr" rid="B30">2022</xref>), we design our experiments for them. For each experiment, in order to explore the impact of model capacity on the performance, we construct two models with high and low capacities, respectively, denoted by the superscription &#x0002B; and -. We choose the architectures for the models with low capacity by trying out different complexities and depths and picking the one with the lowest capacity with the same level of performance. Then for the models with high capacity, we simply increase the numbers of kernels in each layer of the models with low capacity.</p>
<p>Detailed descriptions of all the experiments are reported below, and a summary of the experiments can be found in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Criteria and properties of experiments.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Experiment</bold></th>
<th valign="top" align="center"><bold><italic>G</italic></bold></th>
<th valign="top" align="center"><bold>&#x00023;Params</bold></th>
<th valign="top" align="center"><bold>&#x00023;Epochs</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="4"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>-</sup></td>
<td valign="top" align="center" rowspan="4">&#x1D54B;<sup>3</sup></td>
<td valign="top" align="center" rowspan="2">13,539</td>
<td valign="top" align="center">34</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>-</sup></td>
<td valign="top" align="center">66</td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>&#x0002B;</sup></td>
<td valign="top" align="center" rowspan="2">972,694</td>
<td valign="top" align="center">19</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">67</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="4"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>-</sup></td>
<td valign="top" align="center" rowspan="4"><italic>SO</italic>(3)</td>
<td valign="top" align="center" rowspan="2">286</td>
<td valign="top" align="center">31</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>-</sup></td>
<td valign="top" align="center">45</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>&#x0002B;</sup></td>
<td valign="top" align="center" rowspan="2">2,104</td>
<td valign="top" align="center">31</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">54</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>-</sup></td>
<td valign="top" align="center" rowspan="4">&#x1D54B;<sup>3</sup>&#x000D7;<italic>SO</italic>(3)</td>
<td valign="top" align="center" rowspan="2">2,514</td>
<td valign="top" align="center">41</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>-</sup></td>
<td valign="top" align="center">80</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>&#x0002B;</sup></td>
<td valign="top" align="center" rowspan="2">59,914</td>
<td valign="top" align="center">15</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">54</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>-</sup></td>
<td valign="top" align="center" rowspan="4"><italic>SE</italic>(3)&#x0002A;</td>
<td valign="top" align="center" rowspan="2">2,514</td>
<td valign="top" align="center">41</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>-</sup></td>
<td valign="top" align="center">49</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>&#x0002B;</sup></td>
<td valign="top" align="center" rowspan="2">59,914</td>
<td valign="top" align="center">15</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">48</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>-</sup></td>
<td valign="top" align="center" rowspan="4"><italic>SE</italic>(3)</td>
<td valign="top" align="center" rowspan="2">2,514</td>
<td valign="top" align="center">41</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>-</sup></td>
<td valign="top" align="center">86</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>&#x0002B;</sup></td>
<td valign="top" align="center" rowspan="2">59,914</td>
<td valign="top" align="center">15</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">42</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p><italic>SE</italic>(3)&#x0002A; indicates the rotations in the spatial part are only a part of the rotations used in the spherical part.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.2 Ablation study</title>
<sec>
<title>4.2.1 &#x1D54B;<sup>3</sup>-Classical CNN</title>
<p>The architecture we use is <italic>ReLU</italic>(&#x0211D;<sup>3</sup> conv)&#x02212;<italic>ReLU</italic>(&#x0211D;<sup>3</sup>conv)&#x02212;<italic>ReLU</italic>(&#x0211D;<sup>3</sup>conv)&#x02212;FC with network setups of a low capacity and a high capacity. FC here is a fully connected layer. We label the small network (90 &#x02212; 5&#x02212;5 &#x02212; 5&#x02212;4) Classical<sup>-</sup> and the big network (90 &#x02212; 120 &#x02212; 120 &#x02212; 90 &#x02212; 4) Classical<sup>&#x0002B;</sup>.</p>
</sec>
<sec>
<title>4.2.2 <italic>SO</italic>(3)-Baseline</title>
<p>In the experiments, we use the <italic>ReLU</italic>(lift) &#x02212;<italic>ReLU</italic>(gconv)&#x02212;project&#x02212;FC architecture as was used in Liu et al. (<xref ref-type="bibr" rid="B29">2021</xref>) but with true <italic>SO</italic>(3)-convolution. The projection layer takes the maximum of the five rotations to collapse the function back to the sphere. We experimented various sizes of the network (10 &#x02212; 20&#x02212;<italic>proj</italic>.&#x02212;4 and 20 &#x02212; 40&#x02212;<italic>proj</italic>.&#x02212;4), in addition to the setup used in Liu et al. (<xref ref-type="bibr" rid="B29">2021</xref>) (1 &#x02212; 5&#x02212;<italic>proj</italic>.&#x02212;4). The network that has the biggest size did not seem to improve the second biggest one; thus, we omit it in this paper. Based on the size of the experiments, we call the small network Baseline<sup>-</sup> and the big network Baseline<sup>&#x0002B;</sup>.</p>
</sec>
<sec>
<title>4.2.3 &#x1D54B;<sup>3</sup> &#x000D7; <italic>SO</italic>(3)-OursDecoupled</title>
<p>We use the architecture <italic>ReLU</italic>(lift)&#x02212;<italic>ReLU</italic>(gconv)&#x02212;<italic>ReLU</italic>(gconv)&#x02212;<italic>ReLU</italic>(gconv) &#x02212;project&#x02212;FC. Using separability discussed in Section 3.1.4, a convolution layer (including lifting) is split into two, and ReLU activation is added between separable layers as well. An illustration of the architecture can be found in <xref ref-type="fig" rid="F4">Figure 4</xref>.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Architecture of the network with group action &#x1D54B;<sup>3</sup> &#x000D7; <italic>SO</italic>(3). Each block is a convolutional layer split into two separable layers. The vertical arrows in each block show the separable convolutions. First, the spherical convolution is applied, followed by the spatial convolution. The last block before the FC layer is equivalent to a single <italic>S</italic><sup>2</sup>-layer as explained in Section 4.1. Illustrations of ReLU actions are omitted for visualization simplicity.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0004.tif"/>
</fig>
<p>We again experiment with two sizes of the network - a small one and a big one. The small network has 5 &#x02212; 5&#x02212;5 &#x02212; 5&#x02212;5 &#x02212; 5&#x02212;5&#x02212;<italic>proj</italic>.&#x02212;4 kernels for each layer, while the big network has 10 &#x02212; 20 &#x02212; 20 &#x02212; 40 &#x02212; 40 &#x02212; 20 &#x02212; 10&#x02212;<italic>proj</italic>.&#x02212;4. We label them OursDecoupled<sup>-</sup> and OursDecoupled<sup>&#x0002B;</sup>.</p>
</sec>
<sec>
<title>4.2.4 <italic>SE</italic>(3)-ours</title>
<p>Here too we use the separable setup described in Section 3.1.4. Thus, a layer is again split into two layers - an <italic>S</italic><sup>2</sup>-layer and an &#x0211D;<sup>3</sup>-layer, both for lifting and group convolution. The <italic>S</italic><sup>2</sup>-layer is defined as shown in <xref ref-type="fig" rid="F1">Figures 1A</xref>, <xref ref-type="fig" rid="F1">B</xref>. We rotate the &#x0211D;<sup>3</sup> kernels and the <italic>S</italic><sup>2</sup> kernels using the same actions. The rotational actions of the kernels can be represented by 60 rotation matrices and is equivalent to the discretization of the <italic>SO</italic>(3) rotation group using the icosahedral symmetry group, as shown in <xref ref-type="fig" rid="F1">Figure 1C</xref>. As in Section 4.2.3, we use the <italic>ReLU</italic>(lift)&#x02212;<italic>ReLU</italic>(gconv) &#x02212;<italic>ReLU</italic>(gconv)&#x02212;<italic>ReLU</italic>(gconv)&#x02212;project&#x02212;FC architecture. After the separation of the layers, the illustration is showcased in <xref ref-type="fig" rid="F5">Figure 5</xref>. As in Section 4.2.3, ReLU activations are added between separable layers as well.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Architecture of the network with group action <italic>SE</italic>(3).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0005.tif"/>
</fig>
<p>In addition, we intend to explore the impact of the equivariance we imposed in &#x0211D;<sup>3</sup> in this section. As was explained above, we align the rotations of the &#x0211D;<sup>3</sup> kernel with the ways the <italic>S</italic><sup>2</sup> kernel moved on the sphere, which is discretized by the 60 rotation symmetries of an icosahedron. At a vertex <italic>x</italic><sub><italic>i</italic></sub>, <italic>i</italic>&#x02208;1, &#x02026;, 12 of an icosahedron, there exists a stabilizer <italic>SO</italic>(3)<sub><italic>x</italic><sub><italic>i</italic></sub></sub> discretized by 5 equally divided rotations that keep <italic>x</italic><sub><italic>i</italic></sub> unchanged. Therefore, we also experiment a partial equivariance in the &#x0211D;<sup>3</sup> roto-translational convolution. This means at each vertex <italic>x</italic><sub><italic>i</italic></sub> of the icosahedron, we only take 1 out of the 5 rotations that discretized <italic>SO</italic>(3)<sub><italic>x</italic><sub><italic>i</italic></sub></sub> instead of using all of them to rotate the spatial kernel. Note that the partially equivariant models are only fully <italic>SE</italic>(3)-equivariant when the kernels have a subgroup <italic>SO</italic>(2) symmetry in them (Bekkers, <xref ref-type="bibr" rid="B6">2019</xref>; Thm 1), which we do not impose and thus equivariance is not guaranteed.</p>
<p>Again, we experiment with two sizes of the network with 5 &#x02212; 5&#x02212;5 &#x02212; 5&#x02212;5 &#x02212; 5&#x02212;5&#x02212;<italic>proj</italic>.&#x02212;4 and 10 &#x02212; 20 &#x02212; 20 &#x02212; 40 &#x02212; 40 &#x02212; 20 &#x02212; 10&#x02212;<italic>proj</italic>.&#x02212;4 kernels, respectively. Therefore, we generate four experiments for this section: OursFull<sup>-</sup>, OursPart<sup>-</sup>, OursFull<sup>&#x0002B;</sup>, and OursPart<sup>&#x0002B;</sup>.</p>
</sec>
<sec>
<title>4.2.5 Data augmentation experiments</title>
<p>To validate the robustness of GCNNs against data variation modeled by group actions, we train all the proposed models with augmented data as well. Each data sample (grid of 7<sup>3</sup> or 1<sup>3</sup>) is randomly rotated on the fly before being fed into the model. To prevent interpolation, the rotations used to transform the data are sampled from a octohedral symmetry group. For DWI data that have directional signals in each voxel, the directions of the signals (<italic>b</italic>-vectors) in each voxel rotate with the voxel grid. In order to guarantee the signal values in each voxel are from the same orientations after augmentation, we interpolate the function values at the orientations-of-interest using the rotated <italic>b</italic>-vectors. Therefore, for Type 1 discretization, we interpolate function values at the original <italic>b</italic>-vectors, and for Type 2 discretization, we interpolate at the pre-defined icosahedron as demonstrated above.</p>
</sec>
</sec>
<sec>
<title>4.3 Results</title>
<p>As was done in Liu et al. (<xref ref-type="bibr" rid="B29">2021</xref>), we trained all networks using <bold>1</bold> scan, validated using <bold>1</bold> scan, and tested using <bold>50</bold> scans. We evaluate the accuracies and Dice scores of the classification of the four regions, respectively, and the overall classification accuracy across all test scans. We have also tried training models with more scans (5 or 10); it does not seem to improve the results significantly. Therefore, we choose to use 1 scan for training. For each class, the accuracy is calculated by <inline-formula><mml:math id="M36"><mml:mfrac><mml:mrow><mml:mi>&#x00023;</mml:mi><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x00023;</mml:mi><mml:mi>C</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>S</mml:mi><mml:mi>a</mml:mi><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula>, and the Dice score is calculated by <inline-formula><mml:math id="M37"><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula> for the class. The overall accuracy is calculated by <inline-formula><mml:math id="M38"><mml:mfrac><mml:mrow><mml:mi>&#x00023;</mml:mi><mml:mi>C</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>&#x00023;</mml:mi><mml:mi>A</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mi>S</mml:mi><mml:mi>a</mml:mi><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:mfrac></mml:math></inline-formula>.</p>
<p>We trained all models until they converge and before overfitting; thus, models of different capacities and different setups are stopped at different epochs. Each model is trained with both original data and augmented data. Details can be found in <xref ref-type="table" rid="T2">Table 2</xref>.</p>
<p>The Dice scores and accuracies of models of low capacity can be found in <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref>, while the Dice scores and accuracies of models of high capacity can be found in <xref ref-type="table" rid="T5">Tables 5</xref>, <xref ref-type="table" rid="T6">6</xref>. The numbers shown in all the tables are the average value and standard deviation across 50 test scans. Examples of predictions compared with the ground truth can be found in <xref ref-type="fig" rid="F6">Figure 6A</xref>.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Statistics of dice scores from experiments using models of low capacity.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="5"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>-</sup></td>
<td valign="top" align="center">0.756 &#x000B1; 0.07</td>
<td valign="top" align="center">0.376 &#x000B1; 0.043</td>
<td valign="top" align="center">0.834 &#x000B1; 0.011</td>
<td valign="top" align="center">0.839 &#x000B1; 0.02</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>-</sup></td>
<td valign="top" align="center">0.625 &#x000B1; 0.11</td>
<td valign="top" align="center">0.128 &#x000B1; 0.021</td>
<td valign="top" align="center">0.77 &#x000B1; 0.017</td>
<td valign="top" align="center">0.806 &#x000B1; 0.017</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="5"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>-</sup></td>
<td valign="top" align="center">0.75 &#x000B1; 0.073</td>
<td valign="top" align="center">0.185 &#x000B1; 0.04</td>
<td valign="top" align="center">0.801 &#x000B1; 0.012</td>
<td valign="top" align="center">0.83 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>-</sup></td>
<td valign="top" align="center">0.741 &#x000B1; 0.074</td>
<td valign="top" align="center">0.232 &#x000B1; 0.048</td>
<td valign="top" align="center">0.805 &#x000B1; 0.014</td>
<td valign="top" align="center">0.835 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>-</sup></td>
<td valign="top" align="center"><bold>0.817</bold>&#x000B1;0.051</td>
<td valign="top" align="center"><bold>0.705</bold>&#x000B1;0.033</td>
<td valign="top" align="center"><bold>0.867</bold>&#x000B1;0.009</td>
<td valign="top" align="center"><bold>0.909</bold>&#x000B1;0.007</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>-</sup></td>
<td valign="top" align="center">0.775 &#x000B1; 0.063</td>
<td valign="top" align="center">0.639 &#x000B1; 0.038</td>
<td valign="top" align="center">0.851 &#x000B1; 0.01</td>
<td valign="top" align="center">0.886 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>-</sup></td>
<td valign="top" align="center">0.807 &#x000B1; 0.048</td>
<td valign="top" align="center">0.658 &#x000B1; 0.037</td>
<td valign="top" align="center">0.865 &#x000B1; 0.009</td>
<td valign="top" align="center">0.899 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>-</sup></td>
<td valign="top" align="center">0.78 &#x000B1; 0.06</td>
<td valign="top" align="center">0.643 &#x000B1; 0.037</td>
<td valign="top" align="center">0.849 &#x000B1; 0.01</td>
<td valign="top" align="center">0.886 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>-</sup></td>
<td valign="top" align="center">0.769 &#x000B1; 0.06</td>
<td valign="top" align="center">0.621 &#x000B1; 0.038</td>
<td valign="top" align="center">0.854 &#x000B1; 0.01</td>
<td valign="top" align="center">0.891 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>-</sup></td>
<td valign="top" align="center">0.772 &#x000B1; 0.061</td>
<td valign="top" align="center">0.637 &#x000B1; 0.037</td>
<td valign="top" align="center">0.846 &#x000B1; 0.01</td>
<td valign="top" align="center">0.884 &#x000B1; 0.009</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Statistics of classification accuracy from all experiments using models of low capacity.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
<th valign="top" align="center"><bold>Overall</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>-</sup></td>
<td valign="top" align="center">0.792 &#x000B1; 0.08</td>
<td valign="top" align="center">0.415 &#x000B1; 0.053</td>
<td valign="top" align="center"><bold>0.879</bold>&#x000B1;0.024</td>
<td valign="top" align="center">0.789 &#x000B1; 0.034</td>
<td valign="top" align="center">0.806 &#x000B1; 0.017</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>-</sup></td>
<td valign="top" align="center">0.662 &#x000B1; 0.105</td>
<td valign="top" align="center">0.088 &#x000B1; 0.017</td>
<td valign="top" align="center">0.808 &#x000B1; 0.042</td>
<td valign="top" align="center">0.801 &#x000B1; 0.039</td>
<td valign="top" align="center">0.761 &#x000B1; 0.014</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>-</sup></td>
<td valign="top" align="center">0.742 &#x000B1; 0.082</td>
<td valign="top" align="center">0.145 &#x000B1; 0.04</td>
<td valign="top" align="center">0.804 &#x000B1; 0.024</td>
<td valign="top" align="center">0.85 &#x000B1; 0.016</td>
<td valign="top" align="center">0.788 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>-</sup></td>
<td valign="top" align="center">0.785 &#x000B1; 0.074</td>
<td valign="top" align="center">0.202 &#x000B1; 0.055</td>
<td valign="top" align="center">0.793 &#x000B1; 0.028</td>
<td valign="top" align="center">0.858 &#x000B1; 0.018</td>
<td valign="top" align="center">0.791 &#x000B1; 0.012</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>-</sup></td>
<td valign="top" align="center"><bold>0.844</bold>&#x000B1;0.061</td>
<td valign="top" align="center">0.741 &#x000B1; 0.033</td>
<td valign="top" align="center">0.833 &#x000B1; 0.02</td>
<td valign="top" align="center"><bold>0.934</bold>&#x000B1;0.013</td>
<td valign="top" align="center"><bold>0.878</bold>&#x000B1;0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>-</sup></td>
<td valign="top" align="center">0.769 &#x000B1; 0.087</td>
<td valign="top" align="center">0.716 &#x000B1; 0.04</td>
<td valign="top" align="center">0.854 &#x000B1; 0.023</td>
<td valign="top" align="center">0.87 &#x000B1; 0.023</td>
<td valign="top" align="center">0.853 &#x000B1; 0.01</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>-</sup></td>
<td valign="top" align="center">0.787 &#x000B1; 0.068</td>
<td valign="top" align="center">0.717 &#x000B1; 0.032</td>
<td valign="top" align="center">0.848 &#x000B1; 0.019</td>
<td valign="top" align="center">0.906 &#x000B1; 0.016</td>
<td valign="top" align="center">0.868 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>-</sup></td>
<td valign="top" align="center">0.772 &#x000B1; 0.081</td>
<td valign="top" align="center"><bold>0.752</bold>&#x000B1;0.036</td>
<td valign="top" align="center">0.848 &#x000B1; 0.021</td>
<td valign="top" align="center">0.87 &#x000B1; 0.022</td>
<td valign="top" align="center">0.852 &#x000B1; 0.01</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>-</sup></td>
<td valign="top" align="center">0.81 &#x000B1; 0.065</td>
<td valign="top" align="center">0.692 &#x000B1; 0.029</td>
<td valign="top" align="center">0.857 &#x000B1; 0.022</td>
<td valign="top" align="center">0.874 &#x000B1; 0.019</td>
<td valign="top" align="center">0.856 &#x000B1; 0.01</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>-</sup></td>
<td valign="top" align="center">0.783 &#x000B1; 0.077</td>
<td valign="top" align="center">0.711 &#x000B1; 0.054</td>
<td valign="top" align="center">0.855 &#x000B1; 0.023</td>
<td valign="top" align="center">0.864 &#x000B1; 0.021</td>
<td valign="top" align="center">0.85 &#x000B1; 0.01</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Statistics of dice scores from experiments using models of high capacity.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="5"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.804 &#x000B1; 0.053</td>
<td valign="top" align="center">0.583 &#x000B1; 0.036</td>
<td valign="top" align="center">0.856 &#x000B1; 0.011</td>
<td valign="top" align="center">0.893 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.752 &#x000B1; 0.069</td>
<td valign="top" align="center">0.407 &#x000B1; 0.044</td>
<td valign="top" align="center">0.828 &#x000B1; 0.011</td>
<td valign="top" align="center">0.849 &#x000B1; 0.017</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="5"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.754 &#x000B1; 0.069</td>
<td valign="top" align="center">0.334 &#x000B1; 0.037</td>
<td valign="top" align="center">0.805 &#x000B1; 0.013</td>
<td valign="top" align="center">0.841 &#x000B1; 0.012</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.748 &#x000B1; 0.072</td>
<td valign="top" align="center">0.311 &#x000B1; 0.037</td>
<td valign="top" align="center">0.796 &#x000B1; 0.016</td>
<td valign="top" align="center">0.845 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.827 &#x000B1; 0.047</td>
<td valign="top" align="center">0.716 &#x000B1; 0.044</td>
<td valign="top" align="center">0.878 &#x000B1; 0.009</td>
<td valign="top" align="center">0.903 &#x000B1; 0.01</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.79 &#x000B1; 0.053</td>
<td valign="top" align="center">0.721 &#x000B1; 0.033</td>
<td valign="top" align="center">0.87 &#x000B1; 0.009</td>
<td valign="top" align="center">0.902 &#x000B1; 0.007</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>&#x0002B;</sup></td>
<td valign="top" align="center"><bold>0.834</bold>&#x000B1;0.045</td>
<td valign="top" align="center"><bold>0.752</bold>&#x000B1;0.034</td>
<td valign="top" align="center"><bold>0.878</bold>&#x000B1;0.009</td>
<td valign="top" align="center"><bold>0.914</bold>&#x000B1;0.007</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.789 &#x000B1; 0.059</td>
<td valign="top" align="center">0.736 &#x000B1; 0.035</td>
<td valign="top" align="center">0.872 &#x000B1; 0.009</td>
<td valign="top" align="center">0.902 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.788 &#x000B1; 0.05</td>
<td valign="top" align="center">0.746 &#x000B1; 0.034</td>
<td valign="top" align="center">0.877 &#x000B1; 0.008</td>
<td valign="top" align="center">0.909 &#x000B1; 0.006</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.792 &#x000B1; 0.051</td>
<td valign="top" align="center">0.737 &#x000B1; 0.031</td>
<td valign="top" align="center">0.873 &#x000B1; 0.009</td>
<td valign="top" align="center">0.907 &#x000B1; 0.007</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Statistics of classification accuracy from all experiments using models of high capacity.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
<th valign="top" align="center"><bold>Overall</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.815 &#x000B1; 0.061</td>
<td valign="top" align="center">0.702 &#x000B1; 0.026</td>
<td valign="top" align="center">0.834 &#x000B1; 0.022</td>
<td valign="top" align="center">0.89 &#x000B1; 0.011</td>
<td valign="top" align="center">0.854 &#x000B1; 0.012</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.687 &#x000B1; 0.088</td>
<td valign="top" align="center">0.42 &#x000B1; 0.04</td>
<td valign="top" align="center">0.863 &#x000B1; 0.031</td>
<td valign="top" align="center">0.818 &#x000B1; 0.038</td>
<td valign="top" align="center">0.812 &#x000B1; 0.015</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.778 &#x000B1; 0.07</td>
<td valign="top" align="center">0.379 &#x000B1; 0.065</td>
<td valign="top" align="center">0.784 &#x000B1; 0.024</td>
<td valign="top" align="center">0.848 &#x000B1; 0.02</td>
<td valign="top" align="center">0.792 &#x000B1; 0.013</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.776 &#x000B1; 0.076</td>
<td valign="top" align="center">0.351 &#x000B1; 0.067</td>
<td valign="top" align="center">0.749 &#x000B1; 0.029</td>
<td valign="top" align="center">0.875 &#x000B1; 0.017</td>
<td valign="top" align="center">0.789 &#x000B1; 0.014</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.865 &#x000B1; 0.061</td>
<td valign="top" align="center">0.783 &#x000B1; 0.035</td>
<td valign="top" align="center">0.867 &#x000B1; 0.017</td>
<td valign="top" align="center">0.902 &#x000B1; 0.019</td>
<td valign="top" align="center">0.879 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.821 &#x000B1; 0.066</td>
<td valign="top" align="center">0.759 &#x000B1; 0.052</td>
<td valign="top" align="center">0.876 &#x000B1; 0.02</td>
<td valign="top" align="center">0.891 &#x000B1; 0.018</td>
<td valign="top" align="center">0.876 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.819 &#x000B1; 0.065</td>
<td valign="top" align="center">0.816 &#x000B1; 0.031</td>
<td valign="top" align="center">0.845 &#x000B1; 0.019</td>
<td valign="top" align="center"><bold>0.936</bold>&#x000B1;0.011</td>
<td valign="top" align="center"><bold>0.888</bold>&#x000B1;0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.756 &#x000B1; 0.084</td>
<td valign="top" align="center">0.816 &#x000B1; 0.033</td>
<td valign="top" align="center"><bold>0.876</bold>&#x000B1;0.017</td>
<td valign="top" align="center">0.888 &#x000B1; 0.017</td>
<td valign="top" align="center">0.877 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>&#x0002B;</sup></td>
<td valign="top" align="center"><bold>0.896</bold>&#x000B1;0.042</td>
<td valign="top" align="center"><bold>0.826</bold>&#x000B1;0.023</td>
<td valign="top" align="center">0.857 &#x000B1; 0.017</td>
<td valign="top" align="center">0.912 &#x000B1; 0.014</td>
<td valign="top" align="center">0.883 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.864 &#x000B1; 0.048</td>
<td valign="top" align="center">0.78 &#x000B1; 0.031</td>
<td valign="top" align="center">0.866 &#x000B1; 0.019</td>
<td valign="top" align="center">0.905 &#x000B1; 0.016</td>
<td valign="top" align="center">0.88 &#x000B1; 0.008</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Examples of predictions. <bold>(A)</bold> shows the predictions from the original test set, and <bold>(B)</bold> shows the predictions from the augmented (rotated) test set. In <bold>(A)</bold>, from left to right are ground-truth, Classical<sup>&#x0002B;</sup>, Baseline<sup>&#x0002B;</sup>, OursDecoupled<sup>&#x0002B;</sup>, OursPart<sup>&#x0002B;</sup>, and OursFull<sup>&#x0002B;</sup>. In <bold>(B)</bold>, from left to right are Classical<sup>&#x0002B;</sup>, Baseline<sup>&#x0002B;</sup>, OursDecoupled<sup>&#x0002B;</sup>, OursPart<sup>&#x0002B;</sup>, and OursFull<sup>&#x0002B;</sup>. The colors of CSF, subcortical, WM, and GM are red, blue, white, and gray, respectively. <bold>(A)</bold> Predictions using original data. <bold>(B)</bold> Predictions using rotated data.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0006.tif"/>
</fig>
<sec>
<title>4.3.1 The impact of data augmentation</title>
<p>As we can see from the <xref ref-type="table" rid="T3">Tables 3</xref>&#x02013;<xref ref-type="table" rid="T6">6</xref>, models trained with augmented data do not perform better than their counterparts trained with just original data, if not worse. Unlike 2D image datasets in the computer vision community that have various backgrounds and objects in their images, the HCP dataset is very uniform; thus, the distribution of the original training data is expected to be the same as the test set data. However, after augmentation, the distribution of the training data changed and it differs from the test data. Therefore, in this case, data augmentation does not help any of the models since the augmentation does not represent the diversity in this dataset. One extreme would be Classical<sup>-</sup> vs. ClassicalAug<sup>-</sup> that can be found in <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref>, the augmented data confused the model in terms of the subcortical region - a somewhat mixture of white and gray matter which is challenging for models to distinguish. Therefore, from now on, if not specified, we mainly discuss the models and results trained without data augmentation.</p>
</sec>
<sec>
<title>4.3.2 The impact of the &#x0211D;<sup>3</sup> spatial component</title>
<p>It is easy to observe that the the Baseline experiments perform worst among all. This is an anticipated outcome since it is usually the case that neighboring information is an essential type of local features.</p>
</sec>
<sec>
<title>4.3.3 Type 1 discretization vs Type 2 discretization</title>
<p>The classical CNNs use Type 1 discretization, while Type 2 discretization is used for the rest of the models. The classical CNNs do not perform as well as models that take into account the spherical geometry with spatial information but performs better than Baseline. However, Classical<sup>-</sup> is not much better than Baseline<sup>&#x0002B;</sup> while having far more parameters to train, and Classical<sup>&#x0002B;</sup> performs even worse than OursDecoupled<sup>-</sup>, OursPart<sup>-</sup>, or OursFull<sup>-</sup>, which have much less training parameters.</p>
<p>The results of the two extreme cases&#x02014;Baseline that only takes into account spherical geometry but ignore any spatial information and Classical that only looks into the spatial part and discards spherical geometry&#x02014;show that the voxel geometry and neighboring voxel correlation can both capture some decent amount of information to deal with the segmentation task, but they both have something that the other one cannot grasp, and combining the spherical geometry and the spatial correlation can boost the performance to a promising extent.</p>
</sec>
<sec>
<title>4.3.4 The impact of adding an &#x0211D;<sup>3</sup> part to baseline</title>
<p>On top of the Baseline, the easiest way to add spatial information to the purely voxel-based framework is what was done in OursDecoupled Section 4.2.3&#x02014;a GCNN on <italic>S</italic><sup>2</sup> to learn the geometric signals in individual signals and a regular classical CNN to take into account the local spatial information. We can see from the results that this setup immediately boosted the performance compared to the Baseline. We can also see that OursDecoupled<sup>&#x0002B;</sup> performs better than OursDecoupled<sup>-</sup>, for the sake of model capacity.</p>
</sec>
<sec>
<title>4.3.5 The argument for OursFull not performing the best</title>
<p>For models of low capacity, however, we can observe from <xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref> that our proposed method performs worse than OursDecoupled<sup>-</sup>. In addition, for models of high capacity, even though we can see that OursFull<sup>&#x0002B;</sup> and OursPart<sup>&#x0002B;</sup> improve from their low capacity counterparts more than OursDecoupled<sup>&#x0002B;</sup>, OursFull<sup>&#x0002B;</sup> does not perform as well as OursPart<sup>&#x0002B;</sup> as shown in <xref ref-type="table" rid="T5">Tables 5</xref>, <xref ref-type="table" rid="T6">6</xref>. This differs from our expectation since models with full roto-translational equivariance should be more capable of handling variances in data, thus should have better performance. Recall that the HCP dataset (Van Essen et al., <xref ref-type="bibr" rid="B44">2013</xref>) contains scans that are preprocessed and aligned with axes, thus there is little variance in rotation. In this case, enforcing <italic>SE</italic>(3) equivariance in the model can be futile and be even confusing for the model.</p>
<p>To verify this theory, we evaluated all models on the rotated test set. Taking the <italic>N</italic><sup>3</sup> (<italic>N</italic> &#x0003D; 1 for Baseline models and <italic>N</italic> &#x0003D; 7 for the rest) grids of voxels we extracted from the test scans, we randomly rotate each grid using a rotation sampled from the octahedral symmetry group to create a new rotated test set. In this way, we do not need to interpolate while rotating, and the rotations are not aligned with the ones we used in our models to rotate the kernels while still resemble a discretization of the <italic>SO</italic>(3) group. Hence, we have two categories of models as well as two categories of the test set: models trained with original data vs. models trained with augmented data, and original test set vs. the randomly rotated test set.</p>
</sec>
<sec>
<title>4.3.6 Models trained with data augmentation tested with rotated test set</title>
<p>We see that all models trained with augmented training set have very similar performance results to the same models tested with the original test set, and they all perform better in this task than their counterparts trained with the original training set. This checks with our statement in Section 4.3.1 that the consistency of data distributions of the training and test sets boosts test performance. In this case, we used the same kind of rotations while augmenting the training set and test set; therefore, the consistency of data distributions is maintained. However, this can never be guaranteed in real life. We can see this from <xref ref-type="table" rid="T7">Tables 7</xref>&#x02013;<xref ref-type="table" rid="T10">10</xref>.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Statistics of dice scores from experiments using rotated data and models of low capacity.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="5"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>-</sup></td>
<td valign="top" align="center">0.631 &#x000B1; 0.097</td>
<td valign="top" align="center">0.101 &#x000B1; 0.014</td>
<td valign="top" align="center">0.696 &#x000B1; 0.019</td>
<td valign="top" align="center">0.558 &#x000B1; 0.044</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>-</sup></td>
<td valign="top" align="center">0.678 &#x000B1; 0.094</td>
<td valign="top" align="center">0.117 &#x000B1; 0.025</td>
<td valign="top" align="center">0.775 &#x000B1; 0.018</td>
<td valign="top" align="center">0.813 &#x000B1; 0.019</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="5"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>-</sup></td>
<td valign="top" align="center">0.735 &#x000B1; 0.076</td>
<td valign="top" align="center">0.158 &#x000B1; 0.037</td>
<td valign="top" align="center">0.799 &#x000B1; 0.013</td>
<td valign="top" align="center">0.829 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>-</sup></td>
<td valign="top" align="center">0.741 &#x000B1; 0.074</td>
<td valign="top" align="center">0.237 &#x000B1; 0.047</td>
<td valign="top" align="center">0.804 &#x000B1; 0.014</td>
<td valign="top" align="center">0.834 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>-</sup></td>
<td valign="top" align="center">0.708 &#x000B1; 0.073</td>
<td valign="top" align="center">0.531 &#x000B1; 0.033</td>
<td valign="top" align="center">0.801 &#x000B1; 0.012</td>
<td valign="top" align="center">0.851 &#x000B1; 0.006</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>-</sup></td>
<td valign="top" align="center">0.771 &#x000B1; 0.065</td>
<td valign="top" align="center">0.641 &#x000B1; 0.036</td>
<td valign="top" align="center"><bold>0.851</bold>&#x000B1;0.01</td>
<td valign="top" align="center">0.886 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>-</sup></td>
<td valign="top" align="center">0.714 &#x000B1; 0.069</td>
<td valign="top" align="center">0.536 &#x000B1; 0.035</td>
<td valign="top" align="center">0.804 &#x000B1; 0.011</td>
<td valign="top" align="center">0.851 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>-</sup></td>
<td valign="top" align="center"><bold>0.784</bold>&#x000B1;0.059</td>
<td valign="top" align="center"><bold>0.642</bold>&#x000B1;0.036</td>
<td valign="top" align="center">0.849 &#x000B1; 0.01</td>
<td valign="top" align="center"><bold>0.887</bold>&#x000B1;0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>-</sup></td>
<td valign="top" align="center">0.737 &#x000B1; 0.065</td>
<td valign="top" align="center">0.517 &#x000B1; 0.033</td>
<td valign="top" align="center">0.823 &#x000B1; 0.01</td>
<td valign="top" align="center">0.867 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>-</sup></td>
<td valign="top" align="center">0.774 &#x000B1; 0.061</td>
<td valign="top" align="center">0.636 &#x000B1; 0.036</td>
<td valign="top" align="center">0.846 &#x000B1; 0.01</td>
<td valign="top" align="center">0.884 &#x000B1; 0.009</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.3.7 Models trained with original data tested with rotated test set</title>
<p>In this section, only models trained without data augmentation are compared and discussed. For models with both low and high capacity, OursFull models have the best performance among other models. OursFull<sup>-</sup> remains 0.823 accuracy, decreased from 0.856 while OursFull<sup>&#x0002B;</sup> decreased from 0.883 to 0.84. This is illustrated in <xref ref-type="table" rid="T8">Tables 8</xref>, <xref ref-type="table" rid="T10">10</xref>. In terms of Dice scores, OursFull<sup>-</sup> performs the best for all classes but the subcortical class, and OursFull<sup>&#x0002B;</sup> has the best results for <bold>all</bold> classes, as shown in <xref ref-type="table" rid="T7">Tables 7</xref>, <xref ref-type="table" rid="T9">9</xref>.</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Statistics of classification accuracy from experiments using rotated data and models of low capacity.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
<th valign="top" align="center"><bold>Overall</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>-</sup></td>
<td valign="top" align="center">0.643 &#x000B1; 0.106</td>
<td valign="top" align="center">0.24 &#x000B1; 0.047</td>
<td valign="top" align="center">0.767 &#x000B1; 0.051</td>
<td valign="top" align="center">0.421 &#x000B1; 0.048</td>
<td valign="top" align="center">0.563 &#x000B1; 0.023</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>-</sup></td>
<td valign="top" align="center">0.677 &#x000B1; 0.105</td>
<td valign="top" align="center">0.08 &#x000B1; 0.02</td>
<td valign="top" align="center">0.811 &#x000B1; 0.044</td>
<td valign="top" align="center">0.811 &#x000B1; 0.043</td>
<td valign="top" align="center">0.767 &#x000B1; 0.016</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>-</sup></td>
<td valign="top" align="center">0.733 &#x000B1; 0.085</td>
<td valign="top" align="center">0.12 &#x000B1; 0.035</td>
<td valign="top" align="center">0.802 &#x000B1; 0.024</td>
<td valign="top" align="center">0.852 &#x000B1; 0.016</td>
<td valign="top" align="center">0.786 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>-</sup></td>
<td valign="top" align="center">0.786 &#x000B1; 0.074</td>
<td valign="top" align="center">0.21 &#x000B1; 0.057</td>
<td valign="top" align="center">0.793 &#x000B1; 0.029</td>
<td valign="top" align="center">0.856 &#x000B1; 0.018</td>
<td valign="top" align="center">0.79 &#x000B1; 0.012</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>-</sup></td>
<td valign="top" align="center">0.755 &#x000B1; 0.076</td>
<td valign="top" align="center">0.528 &#x000B1; 0.037</td>
<td valign="top" align="center">0.779 &#x000B1; 0.02</td>
<td valign="top" align="center"><bold>0.871</bold>&#x000B1;0.013</td>
<td valign="top" align="center">0.81 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>-</sup></td>
<td valign="top" align="center">0.765 &#x000B1; 0.09</td>
<td valign="top" align="center">0.72 &#x000B1; 0.038</td>
<td valign="top" align="center">0.853 &#x000B1; 0.023</td>
<td valign="top" align="center">0.871 &#x000B1; 0.023</td>
<td valign="top" align="center"><bold>0.853</bold>&#x000B1;0.01</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>-</sup></td>
<td valign="top" align="center">0.69 &#x000B1; 0.084</td>
<td valign="top" align="center">0.599 &#x000B1; 0.033</td>
<td valign="top" align="center">0.791 &#x000B1; 0.02</td>
<td valign="top" align="center">0.852 &#x000B1; 0.018</td>
<td valign="top" align="center">0.809 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>-</sup></td>
<td valign="top" align="center">0.778 &#x000B1; 0.081</td>
<td valign="top" align="center"><bold>0.745</bold>&#x000B1;0.038</td>
<td valign="top" align="center">0.849 &#x000B1; 0.021</td>
<td valign="top" align="center">0.87 &#x000B1; 0.021</td>
<td valign="top" align="center">0.853 &#x000B1; 0.01</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>-</sup></td>
<td valign="top" align="center"><bold>0.79</bold>&#x000B1;0.067</td>
<td valign="top" align="center">0.591 &#x000B1; 0.026</td>
<td valign="top" align="center">0.835 &#x000B1; 0.023</td>
<td valign="top" align="center">0.84 &#x000B1; 0.022</td>
<td valign="top" align="center">0.823 &#x000B1; 0.01</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>-</sup></td>
<td valign="top" align="center">0.785 &#x000B1; 0.077</td>
<td valign="top" align="center">0.707 &#x000B1; 0.053</td>
<td valign="top" align="center"><bold>0.854</bold>&#x000B1;0.023</td>
<td valign="top" align="center">0.865 &#x000B1; 0.021</td>
<td valign="top" align="center">0.85 &#x000B1; 0.01</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T9">
<label>Table 9</label>
<caption><p>Statistics of dice scores from experiments using rotated data and models of high capacity.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="5"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.549 &#x000B1; 0.106</td>
<td valign="top" align="center">0.124 &#x000B1; 0.007</td>
<td valign="top" align="center">0.535 &#x000B1; 0.014</td>
<td valign="top" align="center">0.59 &#x000B1; 0.022</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.768 &#x000B1; 0.066</td>
<td valign="top" align="center">0.445 &#x000B1; 0.038</td>
<td valign="top" align="center">0.82 &#x000B1; 0.015</td>
<td valign="top" align="center">0.857 &#x000B1; 0.014</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="5"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.733 &#x000B1; 0.076</td>
<td valign="top" align="center">0.282 &#x000B1; 0.036</td>
<td valign="top" align="center">0.799 &#x000B1; 0.013</td>
<td valign="top" align="center">0.839 &#x000B1; 0.012</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.748 &#x000B1; 0.072</td>
<td valign="top" align="center">0.311 &#x000B1; 0.037</td>
<td valign="top" align="center">0.796 &#x000B1; 0.016</td>
<td valign="top" align="center">0.844 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.702 &#x000B1; 0.075</td>
<td valign="top" align="center">0.497 &#x000B1; 0.037</td>
<td valign="top" align="center">0.8 &#x000B1; 0.011</td>
<td valign="top" align="center">0.829 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center"><bold>0.794</bold>&#x000B1;0.054</td>
<td valign="top" align="center">0.723 &#x000B1; 0.033</td>
<td valign="top" align="center">0.87 &#x000B1; 0.009</td>
<td valign="top" align="center">0.902 &#x000B1; 0.007</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.734 &#x000B1; 0.063</td>
<td valign="top" align="center">0.58 &#x000B1; 0.033</td>
<td valign="top" align="center">0.806 &#x000B1; 0.011</td>
<td valign="top" align="center">0.862 &#x000B1; 0.006</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.791 &#x000B1; 0.058</td>
<td valign="top" align="center"><bold>0.736</bold>&#x000B1;0.034</td>
<td valign="top" align="center">0.872 &#x000B1; 0.009</td>
<td valign="top" align="center">0.901 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.74 &#x000B1; 0.06</td>
<td valign="top" align="center">0.604 &#x000B1; 0.034</td>
<td valign="top" align="center">0.835 &#x000B1; 0.01</td>
<td valign="top" align="center">0.877 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.79 &#x000B1; 0.051</td>
<td valign="top" align="center">0.735 &#x000B1; 0.03</td>
<td valign="top" align="center"><bold>0.872</bold>&#x000B1;0.009</td>
<td valign="top" align="center"><bold>0.907</bold>&#x000B1;0.007</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T10">
<label>Table 10</label>
<caption><p>Statistics of classification accuracy from experiments using rotated data and models of high capacity.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
<th valign="top" align="center"><bold>Overall</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup></td>
</tr>
<tr>
<td valign="top" align="left">Classical<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.632 &#x000B1; 0.097</td>
<td valign="top" align="center">0.452 &#x000B1; 0.02</td>
<td valign="top" align="center">0.434 &#x000B1; 0.018</td>
<td valign="top" align="center">0.5 &#x000B1; 0.03</td>
<td valign="top" align="center">0.471 &#x000B1; 0.015</td>
</tr>
<tr>
<td valign="top" align="left">ClassicalAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.71 &#x000B1; 0.088</td>
<td valign="top" align="center">0.517 &#x000B1; 0.033</td>
<td valign="top" align="center">0.811 &#x000B1; 0.038</td>
<td valign="top" align="center">0.85 &#x000B1; 0.034</td>
<td valign="top" align="center">0.812 &#x000B1; 0.015</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;</td>
</tr>
<tr>
<td valign="top" align="left">Baseline<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.769 &#x000B1; 0.074</td>
<td valign="top" align="center">0.307 &#x000B1; 0.059</td>
<td valign="top" align="center">0.782 &#x000B1; 0.024</td>
<td valign="top" align="center">0.846 &#x000B1; 0.02</td>
<td valign="top" align="center">0.786 &#x000B1; 0.013</td>
</tr>
<tr>
<td valign="top" align="left">BaselineAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.776 &#x000B1; 0.076</td>
<td valign="top" align="center">0.356 &#x000B1; 0.068</td>
<td valign="top" align="center">0.749 &#x000B1; 0.029</td>
<td valign="top" align="center">0.873 &#x000B1; 0.017</td>
<td valign="top" align="center">0.788 &#x000B1; 0.014</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupled<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.756 &#x000B1; 0.082</td>
<td valign="top" align="center">0.597 &#x000B1; 0.034</td>
<td valign="top" align="center">0.797 &#x000B1; 0.019</td>
<td valign="top" align="center">0.81 &#x000B1; 0.019</td>
<td valign="top" align="center">0.791 &#x000B1; 0.01</td>
</tr>
<tr>
<td valign="top" align="left">OursDecoupledAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.819 &#x000B1; 0.067</td>
<td valign="top" align="center">0.761 &#x000B1; 0.051</td>
<td valign="top" align="center">0.876 &#x000B1; 0.019</td>
<td valign="top" align="center">0.891 &#x000B1; 0.018</td>
<td valign="top" align="center">0.876 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursPart<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.716 &#x000B1; 0.078</td>
<td valign="top" align="center">0.635 &#x000B1; 0.033</td>
<td valign="top" align="center">0.78 &#x000B1; 0.021</td>
<td valign="top" align="center">0.876 &#x000B1; 0.012</td>
<td valign="top" align="center">0.819 &#x000B1; 0.008</td>
</tr>
<tr>
<td valign="top" align="left">OursPartAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.762 &#x000B1; 0.085</td>
<td valign="top" align="center"><bold>0.811</bold>&#x000B1;0.032</td>
<td valign="top" align="center"><bold>0.878</bold>&#x000B1;0.018</td>
<td valign="top" align="center">0.886 &#x000B1; 0.017</td>
<td valign="top" align="center">0.877 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursFull<sup>&#x0002B;</sup></td>
<td valign="top" align="center"><bold>0.88</bold>&#x000B1;0.048</td>
<td valign="top" align="center">0.659 &#x000B1; 0.028</td>
<td valign="top" align="center">0.83 &#x000B1; 0.019</td>
<td valign="top" align="center">0.868 &#x000B1; 0.018</td>
<td valign="top" align="center">0.84 &#x000B1; 0.009</td>
</tr>
<tr>
<td valign="top" align="left">OursFullAug<sup>&#x0002B;</sup></td>
<td valign="top" align="center">0.862 &#x000B1; 0.049</td>
<td valign="top" align="center">0.78 &#x000B1; 0.031</td>
<td valign="top" align="center">0.865 &#x000B1; 0.019</td>
<td valign="top" align="center"><bold>0.904</bold>&#x000B1;0.016</td>
<td valign="top" align="center"><bold>0.88</bold>&#x000B1;0.008</td>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<p>It is worth noticing that Baseline models almost do not suffer from performance drop while applied with rotated data. It is an <italic>SO</italic>(3)-network that preserves rotational equivariance on <italic>S</italic><sup>2</sup>. For a single-voxel input, the network is very resistant to variations, but the performance of this model is limited due to the lack of spatial interaction and thus in general worse than models with spatial interplay.</p>
<p>Examples of predictions using the rotated test set can be found in <xref ref-type="fig" rid="F6">Figure 6B</xref>. It is easily observed that the classical CNN does not generalize well to the data variation, while models with rotational symmetry (either <italic>SO</italic>(3), &#x1D54B;<sup>3</sup> &#x000D7; <italic>SO</italic>(3), or <italic>SE</italic>(3)) generate better results. However, it is also noticeable that for a challenging minority class, subcortical region, OursFull<sup>&#x0002B;</sup> performs better than the others while other models with some rotational equivariance do not predict a concentrated subcortical region. Zoom-in examples can be found in <xref ref-type="fig" rid="F7">Figure 7</xref>. Predictions from Baseline are omitted from <xref ref-type="fig" rid="F7">Figure 7</xref> since it does not have the same level of performance.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Showcases of zoom-in regions from predictions of the rotated test set. For both scan slices presented, from left to right, top to bottom, are the ground truth, prediction from OursDecoupled<sup>&#x0002B;</sup>, OursPart<sup>&#x0002B;</sup>, and OursFull<sup>&#x0002B;</sup>. The colors of different regions are the same as in <xref ref-type="fig" rid="F6">Figure 6</xref>. <bold>(A)</bold> A test scan slice. <bold>(B)</bold> Another test scan slice.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0007.tif"/>
</fig>
<sec>
<title>4.3.7.1 Augmentation in training data vs. augmentation in testing data</title>
<p>We have experimented models trained with both the original training set and augmented training set, and models tested with both the original test set and randomly rotated test set. The random rotations applied to the test set can be seen as augmentation too. As was discussed above, data augmentation changes the distribution of the dataset, which creates inconsistency between the training and testing set. However, augmentation in the training set enables the models to see more data and thus even tested with the original test set, the performance of any model does not go far off, since the model has seen the type of data in the test set. The performance of models trained with data augmentation is worse than that of models trained with the original training set, though, due to the inconsistency of distributions between the training set and test set when only one of them is augmented. <xref ref-type="fig" rid="F8">Figure 8A</xref> shows, for models tested with the original test set only, the decrease of model performance from models trained with the original training set to models trained with data augmentation. The <italic>y</italic>-axis shows the logistic map of the ratio of the performance decrease and is calculated by <inline-formula><mml:math id="M39"><mml:mi>L</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mrow><mml:mi>e</mml:mi></mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:math></inline-formula> with &#x003B1; &#x0003D; 20, <inline-formula><mml:math id="M40"><mml:mi>x</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>u</mml:mi><mml:mi>g</mml:mi><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></inline-formula>, and <italic>C</italic><sub><italic>original</italic></sub> and <italic>C</italic><sub><italic>augmented</italic></sub> are the numbers indicating the performance (in this case, either dice score or accuracy as shown in the figure) of models tested with only the original test set but trained with the original (<italic>C</italic><sub><italic>original</italic></sub>) or augmented (<italic>C</italic><sub><italic>augmented</italic></sub>) training set. We can see from <xref ref-type="fig" rid="F8">Figure 8A</xref> that the performance of the equivariant models we propose decrease less. This shows, from one perspective, the resistance of equivariant models to inconsistency of data distributions between training and testing data. On the other hand, having data augmentation only in the test set becomes a big problem for models without equivariance. <xref ref-type="fig" rid="F8">Figure 8B</xref> shows, for models trained with the original training set only, the performance decrease from models tested with the original test set to those tested with rotated data. The <italic>y</italic>-axis values are calculated the same as the formula above, but the <italic>C</italic><sub><italic>original</italic></sub> and <italic>C</italic><sub><italic>augmented</italic></sub> become the numbers indicating the performance of models trained with the original training set only but tested with the original (<italic>C</italic><sub><italic>original</italic></sub>) or rotated (<italic>C</italic><sub><italic>augmented</italic></sub>) test set. We can see clearly from <xref ref-type="fig" rid="F8">Figure 8B</xref> as well that the performance of classical CNN decreases the most using rotated data, and the decrease of performance goes down when we enforce more spatial equivariance in the model. Baseline models decrease the least, but again, the performance is limited due to the lack of information in &#x0211D;<sup>3</sup>. Furthermore, the <italic>SE</italic>(3)-equivariance is implemented separately for the spatial and spherical parts and is with interpolation in the spatial part; thus, there are some errors introduced to it. Therefore, OursFull models always perform the best when there is variation in the test data.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Logistic map of the ratio of two criteria to evaluate the proposed models. One criterion is for the models trained with augmented data compared to their counterparts trained with original data. For models trained both with original and augmented data, the left figure shows the decrease of test results while trained with data augmentation and tested with the original test set as shown in <xref ref-type="table" rid="T3">Tables 3</xref>&#x02013;<xref ref-type="table" rid="T6">6</xref>. The second criterion is for the models trained with original data only. It is the decrease of performance while tested with rotated data, shown on the right figure. <bold>(A)</bold> Model performance decrease while trained with data augmentation. <bold>(B)</bold> Model performance decrease while applied with rotated test set.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0008.tif"/>
</fig>
</sec>
<sec>
<title>4.3.7.2 Rotational invariance for Type 1 discretization</title>
<p>Furthermore, we have also experimented with networks that have some rotational invariance but in the classical CNN setup - viewing the DWI images as <italic>I</italic>:&#x0211D;<sup>3</sup> &#x02192; &#x0211D;<sup><italic>N</italic></sup>. Taking the classical CNN setup we have in Section 4.2.1, we rotate the CNN kernels in each layer using the same rotations as in Section 4.2.4 to discretize <italic>SO</italic>(3). As was done above, we use the 60 rotations from the icosahedral symmetry group as well as only 12 of them (1 at each rotation axis) to act on the CNN kernels. In each layer, one rotation of the kernel is only convolved with the response of the corresponding rotation from the last layer; thus, this network is in fact 60 (or 12) independent networks, in which they share the same weights of different rotations. At the end, we take the average of the 60 (or 12) responses from all the rotations. With a small trial, we discovered that, as expected, even though this type of network does not perform as well as our spatial-directional GCNN as a whole, the performance decreases little in the full icosahedral group case with 60 rotations when tested with augmented data and decreases more when only a subset (12) of the group is used to rotate the kernels (see <xref ref-type="table" rid="T11">Table 11</xref>).</p>
<table-wrap position="float" id="T11">
<label>Table 11</label>
<caption><p>Augmented CNN tested with original and rotated data.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Rotations</bold></th>
<th valign="top" align="center"><bold>Data type</bold></th>
<th valign="top" align="center"><bold>CSF dice</bold></th>
<th valign="top" align="center"><bold>Subcortical dice</bold></th>
<th valign="top" align="center"><bold>WM dice</bold></th>
<th valign="top" align="center"><bold>GM dice</bold></th>
<th valign="top" align="center"><bold>Overall ACC</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="7">90 &#x02212; 5&#x02212;5 &#x02212; 5&#x02212;<italic>FC</italic><bold>, &#x00023;Param 13539</bold></td>
</tr>
<tr>
<td valign="top" align="left" rowspan="2">Part(12)</td>
<td valign="top" align="center">Original</td>
<td valign="top" align="center">0.798 &#x000B1; 0.058</td>
<td valign="top" align="center">0.425 &#x000B1; 0.052</td>
<td valign="top" align="center">0.843 &#x000B1; 0.01</td>
<td valign="top" align="center">0.875 &#x000B1; 0.01</td>
<td valign="top" align="center">0.838 &#x000B1; 0.011</td>
</tr>
<tr>
<td valign="top" align="center">Rotated</td>
<td valign="top" align="center">0.71 &#x000B1; 0.074</td>
<td valign="top" align="center">0.306 &#x000B1; 0.042</td>
<td valign="top" align="center">0.755 &#x000B1; 0.014</td>
<td valign="top" align="center">0.796 &#x000B1; 0.014</td>
<td valign="top" align="center">0.75 &#x000B1; 0.013</td>
</tr>
<tr>
<td valign="top" align="left" rowspan="2">Full(60)</td>
<td valign="top" align="center">Original</td>
<td valign="top" align="center">0.754 &#x000B1; 0.065</td>
<td valign="top" align="center">0.485 &#x000B1; 0.059</td>
<td valign="top" align="center">0.823 &#x000B1; 0.014</td>
<td valign="top" align="center">0.848 &#x000B1; 0.02</td>
<td valign="top" align="center">0.818 &#x000B1; 0.016</td>
</tr>
<tr>
<td valign="top" align="center">Rotated</td>
<td valign="top" align="center">0.75 &#x000B1; 0.063</td>
<td valign="top" align="center">0.479 &#x000B1; 0.059</td>
<td valign="top" align="center">0.813 &#x000B1; 0.013</td>
<td valign="top" align="center">0.838 &#x000B1; 0.02</td>
<td valign="top" align="center">0.809 &#x000B1; 0.016</td>
</tr></tbody>
</table>
</table-wrap>
<p>This further demonstrates that having rotational equivariance in the model makes it much more robust to variance in the data - which, with no need of explanation, is inevitable when dealing with real-world raw data. Averaging rotational copies of a classical CNN achieves the goal of dealing with variance in data, but for non-linear data such as DWI, for which signals in voxels have some geometric structure, our full <italic>SE</italic>(3)-GCNN provides the best solution.</p>
</sec>
</sec>
</sec>
<sec>
<title>4.4 Comparison to state-of-the-art</title>
<p>We now compare our method to the approach of M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>). They used DWI data with q-space encoding in the diffusion part and the spatial part of the data is referred to as p-space, and these two parts of the data resemble the <italic>S</italic><sup>2</sup> and &#x0211D;<sup>3</sup> spaces in our formulation. We use the <italic>b</italic>-vectors from the HCP dataset as the input to the q-space. In their case, the input of the network is a whole DWI scan, not a series of extracted patches like we do, and we cannot fit an entire HCP scan into the model without exceeding the memory limit of a 24 GB GPU. After discussion and agreement with one of the authors (V. Golkov), we decided to use a modified architecture of their network to get an as fair as possible comparison: (1) we provide their network with patches of the same size as ours (7 &#x000D7; 7 &#x000D7; 7), but with DWI signals that are only normalized by <italic>b</italic>0 instead of interpolated spherical functions in each voxel like we did in our method. (2) The best performing model hyper-parameters they provided in the paper (with 4 and 5 layers in totals) are optimized for receptive fields that are much larger than ours, we use instead their 3-layer network, which has almost the same level of performance. (3) We have also disabled padding in their network to cancel biases introduced in the networks. After 3 <italic>p</italic>-spatial layers, the output of their network without padding has spatial dimensions 1 &#x000D7; 1 &#x000D7; 1. Their method and ours thus perform the same task: voxel-wise classification. We used the Focal Loss (Lin et al., <xref ref-type="bibr" rid="B28">2018</xref>) using the same parameters as all the experiments above. We used the suggested structure of their network with fully connected layers in the radial basis, which reportedly has better performance than ones without them. To make the comparison fair, we use a network whose hyper-parameters are different from what was presented in Liu et al. (<xref ref-type="bibr" rid="B30">2022</xref>) such that the number of trainable parameters is similar to that of M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>).</p>
<sec>
<title>4.4.1 Network architectures</title>
<p>For M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>), we use the 1(<italic>pq</italic>)&#x0002B;1(<italic>q</italic>&#x02212;<italic>reduction</italic>)&#x0002B;2(<italic>p</italic>) layer structure with the <italic>TP</italic>&#x000B1;1 basis presented in their paper and channels (5, 3, 0, 0), (5, 3, 0, 0), (10, 5, 0, 0), (4, 0, 0, 0) as presented in the <xref ref-type="supplementary-material" rid="SM1">Appendix</xref> section E.1 in their paper, except that we changed the output channel to 4 to fit our multiclass classification task and changed the <italic>p</italic>-space kernel sizes to 3 to ensure that the receptive field of the network is 7 &#x000D7; 7 &#x000D7; 7, as we discussed with the author. For our method, we use a <italic>ReLU</italic>(<italic>lift</italic>)&#x02212;<italic>ReLU</italic>(<italic>gconv</italic>)&#x02212;<italic>ReLU</italic>(<italic>gconv</italic>)&#x02212;<italic>project</italic>&#x02212;<italic>FC</italic> architecture such that there are three spatial layers as in M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>). With each layer split into 2, we use 10 &#x02212; 10 &#x02212; 20 &#x02212; 40 &#x02212; 20 &#x02212; 10&#x02212;<italic>proj</italic>.&#x02212;4 as our layer structure such that we have similar numbers of parameters as M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>). Our method has 34964 parameters, while M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>) has 34,781 parameters.</p>
</sec>
<sec>
<title>4.4.2 Results</title>
<p>The results are shown in <xref ref-type="table" rid="T12">Table 12</xref>. We can see that our method performs better than M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>). To test the equivariance of both methods, we again test both models with the randomly rotated test set as presented above, and the results can be found in <xref ref-type="table" rid="T13">Table 13</xref>.</p>
<table-wrap position="float" id="T12">
<label>Table 12</label>
<caption><p>Statistics of results from both our method and M&#x000FC;ller&#x00027;s method.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
<th valign="top" align="center"><bold>Overall</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><bold>Accuracy</bold></td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>0.804</bold>&#x000B1;0.073</td>
<td valign="top" align="center"><bold>0.754</bold>&#x000B1;0.033</td>
<td valign="top" align="center"><bold>0.871</bold>&#x000B1;0.018</td>
<td valign="top" align="center"><bold>0.908</bold>&#x000B1;0.011</td>
<td valign="top" align="center"><bold>0.882</bold>&#x000B1;0.008</td>
</tr>
<tr>
<td valign="top" align="left">M&#x000FC;ller&#x00027;s</td>
<td valign="top" align="center">0.583 &#x000B1; 0.123</td>
<td valign="top" align="center">0.442 &#x000B1; 0.176</td>
<td valign="top" align="center">0.83 &#x000B1; 0.036</td>
<td valign="top" align="center">0.834 &#x000B1; 0.033</td>
<td valign="top" align="center">0.805 &#x000B1; 0.015</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><bold>Dice score</bold></td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>0.799</bold>&#x000B1;0.053</td>
<td valign="top" align="center"><bold>0.722</bold>&#x000B1;0.034</td>
<td valign="top" align="center"><bold>0.877</bold>&#x000B1;0.008</td>
<td valign="top" align="center"><bold>0.908</bold>&#x000B1;0.006</td>
<td/>
</tr>
<tr>
<td valign="top" align="left">M&#x000FC;ller&#x00027;s</td>
<td valign="top" align="center">0.655 &#x000B1; 0.086</td>
<td valign="top" align="center">0.41 &#x000B1; 0.105</td>
<td valign="top" align="center">0.813 &#x000B1; 0.015</td>
<td valign="top" align="center">0.849 &#x000B1; 0.016</td>
<td/>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="T13">
<label>Table 13</label>
<caption><p>Statistics of results from both our method and M&#x000FC;ller&#x00027;s method tested with rotated test set.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><inline-graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-i0002.tif"/></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
<th valign="top" align="center"><bold>Overall</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><bold>Accuracy</bold></td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>0.725</bold>&#x000B1;0.083</td>
<td valign="top" align="center"><bold>0.596</bold>&#x000B1;0.036</td>
<td valign="top" align="center"><bold>0.834</bold>&#x000B1;0.02</td>
<td valign="top" align="center"><bold>0.874</bold>&#x000B1;0.013</td>
<td valign="top" align="center"><bold>0.838</bold>&#x000B1;0.008</td>
</tr>
<tr>
<td valign="top" align="left">M&#x000FC;ller&#x00027;s</td>
<td valign="top" align="center">0.445 &#x000B1; 0.1</td>
<td valign="top" align="center">0.337 &#x000B1; 0.146</td>
<td valign="top" align="center">0.823 &#x000B1; 0.036</td>
<td valign="top" align="center">0.789 &#x000B1; 0.031</td>
<td valign="top" align="center">0.771 &#x000B1; 0.014</td>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><bold>Dice score</bold></td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>0.742</bold>&#x000B1;0.067</td>
<td valign="top" align="center"><bold>0.593</bold>&#x000B1;0.032</td>
<td valign="top" align="center"><bold>0.832</bold>&#x000B1;0.009</td>
<td valign="top" align="center"><bold>0.875</bold>&#x000B1;0.006</td>
<td/>
</tr>
<tr>
<td valign="top" align="left">M&#x000FC;ller&#x00027;s</td>
<td valign="top" align="center">0.426 &#x000B1; 0.055</td>
<td valign="top" align="center">0.343 &#x000B1; 0.104</td>
<td valign="top" align="center">0.787 &#x000B1; 0.015</td>
<td valign="top" align="center">0.813 &#x000B1; 0.015</td>
<td/>
</tr></tbody>
</table>
<table-wrap-foot>
<p>The bold values highlight the maximum value in the column.</p>
</table-wrap-foot>
</table-wrap>
<p>We can see from the numbers that the performance of M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>) does not drop much either while tested with unseen rotated test set, similar to our method. As we can see from <xref ref-type="fig" rid="F9">Figure 9</xref>, overall, M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>) lost less in percentage of the Dice scores of Subcortical, White matter, and overall accuracy but more in CSF Dice score. Both equivariant methods are more resistant to variations in the distributions of the training and test set than the non-equivariant models presented above. Moreover, since the overall performance decrease of M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>) while tested with rotated data is lower than our fully equivariant model, M&#x000FC;ller et al. (<xref ref-type="bibr" rid="B32">2021</xref>) actually has better equivariance than all models we presented even though their prediction accuracies and dice scores are lower.</p>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>Comparison of model performance decrease while applied with rotated test set between our method and M&#x000FC;ller&#x00027;s. The radial axis indicates the decrease, and it is the logistic map of the ratio calculated by the same scheme used in <xref ref-type="fig" rid="F8">Figure 8</xref>.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1369717-g0009.tif"/>
</fig>
</sec>
</sec>
<sec>
<title>4.5 Comparison to non-NN spherical harmonics feature classification</title>
<p>Following the method described in their paper, we extracted spherical harmonic features from each voxel of <italic>b</italic>&#x02212;1000 DWIs and used SVMs for classification. Both one-vs-one and one-vs-all SVM configurations were applied to evaluate their comparative effectiveness in handling multiclass data. To normalize features, we experimented with both standard and min-max normalization methods. The performance of each setup was assessed using accuracy and Dice score metrics, consistent with the evaluation metrics for our proposed method. The results are shown in <xref ref-type="table" rid="T14">Table 14</xref>.</p>
<table-wrap position="float" id="T14">
<label>Table 14</label>
<caption><p>Results for all models from Schnell et al. (<xref ref-type="bibr" rid="B35">2009</xref>) for <italic>b</italic> &#x0003D; 1000.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Norm, metric</bold></th>
<th valign="top" align="center"><bold>CSF</bold></th>
<th valign="top" align="center"><bold>Subcortical</bold></th>
<th valign="top" align="center"><bold>WM</bold></th>
<th valign="top" align="center"><bold>GM</bold></th>
<th valign="top" align="center"><bold>Overall</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><bold>OVO</bold></td>
</tr>
<tr>
<td valign="top" align="left">Standard, ACC</td>
<td valign="top" align="center">0.772 &#x000B1; 0.007</td>
<td valign="top" align="center">0.007 &#x000B1; 0.000</td>
<td valign="top" align="center">0.918 &#x000B1; 0.005</td>
<td valign="top" align="center">0.538 &#x000B1; 0.057</td>
<td valign="top" align="center">0.678 &#x000B1; 0.000</td>
</tr>
<tr>
<td valign="top" align="left">Standard, Dice</td>
<td valign="top" align="center">0.732 &#x000B1; 0.006</td>
<td valign="top" align="center">0.014 &#x000B1; 0.000</td>
<td valign="top" align="center">0.728 &#x000B1; 0.003</td>
<td valign="top" align="center">0.627 &#x000B1; 0.025</td>
<td/>
</tr>
<tr>
<td valign="top" align="left">Minmax, ACC</td>
<td valign="top" align="center">0.785 &#x000B1; 0.006</td>
<td valign="top" align="center">0.003 &#x000B1; 0.000</td>
<td valign="top" align="center">0.906 &#x000B1; 0.007</td>
<td valign="top" align="center">0.576 &#x000B1; 0.075</td>
<td valign="top" align="center">0.692 &#x000B1; 0.000</td>
</tr>
<tr>
<td valign="top" align="left">Minmax, Dice</td>
<td valign="top" align="center">0.729 &#x000B1; 0.007</td>
<td valign="top" align="center">0.005 &#x000B1; 0.000</td>
<td valign="top" align="center">0.739 &#x000B1; 0.004</td>
<td valign="top" align="center">0.642 &#x000B1; 0.034</td>
<td/>
</tr>
<tr style="background-color:#dee1e1">
<td valign="top" align="left" colspan="6"><bold>OVR</bold></td>
</tr>
<tr>
<td valign="top" align="left">Standard, ACC</td>
<td valign="top" align="center">0.695 &#x000B1; 0.011</td>
<td valign="top" align="center">0.000 &#x000B1; 0.000</td>
<td valign="top" align="center">0.920 &#x000B1; 0.004</td>
<td valign="top" align="center">0.570 &#x000B1; 0.044</td>
<td valign="top" align="center">0.693 &#x000B1; 0.000</td>
</tr>
<tr>
<td valign="top" align="left">Standard, Dice</td>
<td valign="top" align="center">0.737 &#x000B1; 0.006</td>
<td valign="top" align="center">0.000 &#x000B1; 0.000</td>
<td valign="top" align="center">0.738 &#x000B1; 0.003</td>
<td valign="top" align="center">0.659 &#x000B1; 0.018</td>
<td/>
</tr>
<tr>
<td valign="top" align="left">Minmax, ACC</td>
<td valign="top" align="center">0.724 &#x000B1; 0.010</td>
<td valign="top" align="center">0.000 &#x000B1; 0.000</td>
<td valign="top" align="center">0.920 &#x000B1; 0.006</td>
<td valign="top" align="center">0.554 &#x000B1; 0.069</td>
<td valign="top" align="center">0.686 &#x000B1; 0.000</td>
</tr>
<tr>
<td valign="top" align="left">Minmax, Dice</td>
<td valign="top" align="center">0.740 &#x000B1; 0.006</td>
<td valign="top" align="center">0.000 &#x000B1; 0.000</td>
<td valign="top" align="center">0.736 &#x000B1; 0.004</td>
<td valign="top" align="center">0.633 &#x000B1; 0.031</td>
<td/>
</tr></tbody>
</table>
<table-wrap-foot>
<p>Both One-vs.-One (OVO) and One-vs.-Rest (OVR) models using Standard and Minmax normalizations are presented. The values shown in the table are the mean and standard deviation of the chosen metrics, and three decimals are used; therefore, some very small values are shown as 0.</p>
</table-wrap-foot>
</table-wrap>
<p>As shown in <xref ref-type="table" rid="T14">Table 14</xref>, the performance of the method from Schnell et al. (<xref ref-type="bibr" rid="B35">2009</xref>) is significantly lower than that of our proposed approach. In particular, for the challenging class&#x02013;the subcortical region&#x02013;the model showed minimal recognition capability. This result is expected as the rotation-invariant features derived independently from individual voxels inherently disregard the spatial relationships among voxels, which undermines model robustness. Furthermore, our <italic>SO</italic>(3) models, which similarly do not incorporate spatial voxel connectivity, nonetheless outperform the method in Schnell et al. (<xref ref-type="bibr" rid="B35">2009</xref>), underscoring the robustness and stability introduced by the equivariant convolutions within our model.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s5">
<title>5 Discussion</title>
<p>The resistance to data variation that has been shown by our fully equivariant network was demonstrated on synthetically augmented data - with 90-degree rotations. Even though this synthetic augmentation did not cost any loss of signals or any interpolation-caused inaccuracy, it is desirable to verify the robustness of more complex group actions in CNNs using data with real-world variations (e.g., subjects scanned in different positions, affine variations in shapes). Acquiring this type of data is another challenge. On the other hand, data augmentation seems to be very robust against the variations in the rotated test set. However, this is because the augmentations applied in the training set and the test set are identical, and they modeled exactly the same distribution in the data. Our proposed equivariant methods deal with inconsistent distributions between the training set and the test set much better, which is usually the case in real world. In addition, our method outperforms (M&#x000FC;ller et al., <xref ref-type="bibr" rid="B32">2021</xref>) with the same amount of information given to the models. Even though both methods show similar resistance to variations in the distributions of the training and test set, our model has a more light-weight implementation using regular group representation with separable kernels. Furthermore, the experiments we conducted using Schnell et al. (<xref ref-type="bibr" rid="B35">2009</xref>) have shown the power of equivariant learning in non-Euclidean spaces. Using rotation-invariant features as in Schnell et al. (<xref ref-type="bibr" rid="B35">2009</xref>) is beneficial in terms of getting consistent features from spherical functions, regardless of the orientation. However, extracting invariant features from the very beginning also discards potentially valuable orientational information that is implicitly embedded in the data, and discarding spatial information completely severely weakens the capability of the model. This is easily shown by the fact that our <italic>SO</italic>(3) models that also discard spatial relationships outperform (Schnell et al., <xref ref-type="bibr" rid="B35">2009</xref>).</p>
<p>In conclusion, we presented a systematic study of GCNNs of various group actions with the application to DWI segmentation. We interpreted images of DWI scans (<italic>I</italic>:&#x0211D;<sup>3</sup>&#x000D7;<italic>S</italic><sup>2</sup> &#x02192; &#x0211D;) as functions in the homogeneous spaces of groups with different complexities of symmetries and provided a detailed analysis of how different levels of complexities of these symmetries impact the performance of the network. It is shown from the models OursDecoupled and OursFull that whether or not more complex transformations should be imposed in the model is not always a clear-cut, since while tested on the original test set, OursDecoupled has a slightly better performance. OursDecoupled incorporates a mathematically well-defined, but physically impossible group action, yet it is computed more cheaply, while OursFull incorporates the SE(3) action, which corresponds to the expected physical transformations of the data. And, under any physically realistic turbulence in the test data resulting in unseen distributions, adding to the model possible transformations of the data (to the limit of their discretizations) provides a more stable performance. Therefore, we emphasize the importance of imposing the full roto-translation transformations in models as it is the kind that appears in the data. From the experiments, we conclude that (1) exploiting the spatial-directional interactions in the data is crucial for efficient learning of the features; (2) incorporating complex group actions of 3D rigid motions&#x02014;SE(3)&#x02014;might not be essential for highly aligned and preprocessed data such as the human connectome project (HCP) (Van Essen et al., <xref ref-type="bibr" rid="B44">2013</xref>), but it shows significantly higher resistance to variations in data. For real-world raw data in which the positions of subjects are not perfectly aligned as in Van Essen et al. (<xref ref-type="bibr" rid="B44">2013</xref>), our proposal shows significant potential.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/<xref ref-type="supplementary-material" rid="SM1">Supplementary material</xref>, further inquiries can be directed to the corresponding author.</p>
</sec>
<sec sec-type="ethics-statement" id="s7">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Connectome Coordination Facility. The studies were conducted in accordance with the local legislation and institutional requirements. Written informed consent for participation was not required from the participants or the participants&#x00027; legal guardians/next of kin in accordance with the national legislation and institutional requirements.</p>
</sec>
<sec sec-type="author-contributions" id="s8">
<title>Author contributions</title>
<p>RL: Conceptualization, Formal analysis, Investigation, Methodology, Project administration, Resources, Software, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. FL: Formal analysis, Methodology, Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. EB: Conceptualization, Methodology, Visualization, Writing &#x02013; review &#x00026; editing. SD: Funding acquisition, Project administration, Resources, Supervision, Writing &#x02013; review &#x00026; editing. KE: Funding acquisition, Resources, Supervision, Writing &#x02013; review &#x00026; editing.</p>
</sec>
<sec sec-type="funding-information" id="s9">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This project has received funding from the European Union&#x00027;s Horizon 2020 research and innovation program under the Marie Sklodowska-Curie grant agreement No. 801199. This study only contains the author&#x00027;s views. The Research Executive Agency and the Commission are not responsible for any use that may be made of the information it contains. Data were provided [in part] by the Human Connectome Project, WU-Minn Consortium (Principal Investigators: David Van Essen and Kamil Ugurbil; 1U54MH091657) funded by the 16 NIH Institutes and Centers that support the NIH Blueprint for Neuroscience Research; and by the McDonnell Center for Systems Neuroscience at Washington University. This project is also partially funded by 3Shape A/S, as well as by the research program VENI (grant number 17290), financed by the Dutch Research Council (NWO).</p>
</sec>
<ack><p>We would like to thank Dr. Vladimir Golkov for his efforts and insights in helping us setting up experiments with their model (M&#x000FC;ller et al., <xref ref-type="bibr" rid="B32">2021</xref>).</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="s11">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/frai.2025.1369717/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/frai.2025.1369717/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/></sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Andrearczyk</surname> <given-names>V.</given-names></name> <name><surname>Fageot</surname> <given-names>J.</given-names></name> <name><surname>Depeursinge</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Local rotation invariance in 3D CNNs</article-title>. <source>Med. Image Analy</source>. <volume>65</volume>:<fpage>101756</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2020.101756</pub-id><pub-id pub-id-type="pmid">32623274</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Aronsson</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Homogeneous vector bundles and &#x1D4A2;-equivariant convolutional neural networks</article-title>. <source>Sampl. Theory Signal Process. Data Anal</source>. 20.</citation>
</ref>
<ref id="B3">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Banerjee</surname> <given-names>M.</given-names></name> <name><surname>Chakraborty</surname> <given-names>R.</given-names></name> <name><surname>Archer</surname> <given-names>D.</given-names></name> <name><surname>Vaillancourt</surname> <given-names>D.</given-names></name> <name><surname>Vemuri</surname> <given-names>B. C.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;DMR-CNN: a CNN tailored for DMR scans with applications to PD calssifiaction,&#x0201D;</article-title> in <source>2019 IEEE 16th International Symposium on Biomedical Imaging (ISBI 2019)</source> (<publisher-loc>Venice</publisher-loc>), <fpage>388</fpage>&#x02013;<lpage>391</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI.2019.8759558</pub-id></citation>
</ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Basser</surname> <given-names>P.</given-names></name> <name><surname>Mattiello</surname> <given-names>J.</given-names></name> <name><surname>LeBihan</surname> <given-names>D.</given-names></name></person-group> (<year>1994</year>). <article-title>MR diffusion tensor spectroscopy and imaging</article-title>. <source>Biophys. J</source>. <volume>66</volume>, <fpage>259</fpage>&#x02013;<lpage>267</lpage>. <pub-id pub-id-type="doi">10.1016/S0006-3495(94)80775-1</pub-id><pub-id pub-id-type="pmid">8130344</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bekkers</surname> <given-names>E.</given-names></name> <name><surname>Veta</surname> <given-names>M. L. M.</given-names></name> <name><surname>Eppenhof</surname> <given-names>K.</given-names></name> <name><surname>Pluim</surname> <given-names>J.</given-names></name> <name><surname>Duits</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>Roto-translation covariant convolutional networks for medical image analysis</article-title>. <source>Proc. MICCAI</source> <volume>2018</volume>, <fpage>440</fpage>&#x02013;<lpage>448</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-00928-1_50</pub-id><pub-id pub-id-type="pmid">33197715</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bekkers</surname> <given-names>E. J.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;B-spline CNNS on lie groups,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source>.</citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boscaini</surname> <given-names>D.</given-names></name> <name><surname>Masci</surname> <given-names>J.</given-names></name> <name><surname>Rodol&#x000E0;</surname> <given-names>E.</given-names></name> <name><surname>Bronstein</surname> <given-names>M.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Learning shape correspondence with anisotropic convolutional neural networks,&#x0201D;</article-title> in <source>30th Annual Conference on Neural Information Processing Systems, NIPS 2016</source>, eds. D. Lee, M. Sugiyama, U. Luxburg, I. Guyon and R. Garnett (Barcelona), <fpage>29</fpage>.<pub-id pub-id-type="pmid">37621734</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bouza</surname> <given-names>J. J.</given-names></name> <name><surname>Yang</surname> <given-names>C.-H.</given-names></name> <name><surname>Vaillancourt</surname> <given-names>D.</given-names></name> <name><surname>Vemuri</surname> <given-names>B. C.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;A higher order manifold-valued convolutional neural network with applications to diffusion MRI processing,&#x0201D;</article-title> in <source>Information Processing in Medical Imaging</source>, A. Feragen, S. Sommer, J. Schnabel, and M. Nielsen (Cham: Springer International Publishing), <fpage>304</fpage>&#x02013;<lpage>317</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Caruyer</surname> <given-names>E.</given-names></name> <name><surname>Verma</surname> <given-names>R.</given-names></name></person-group> (<year>2015</year>). <article-title>On Facilitating the Use of HARDI in Population Studies by Creating Rotation-Invariant Markers</article-title>. <source>Med. Image Anal</source>. <volume>20</volume>, <fpage>87</fpage>&#x02013;<lpage>96</lpage>. <pub-id pub-id-type="doi">10.1016/j.media.2014.10.009</pub-id><pub-id pub-id-type="pmid">25465846</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chakraborty</surname> <given-names>R.</given-names></name> <name><surname>Banerjee</surname> <given-names>M.</given-names></name> <name><surname>Vemuri</surname> <given-names>B.</given-names></name></person-group> (<year>2018a</year>). <article-title>A CNN for homogeneous riemannian manifolds with application to neuroimaging</article-title>. <source>arXiv [Preprint]. arXiv:1805.05487</source>.</citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chakraborty</surname> <given-names>R.</given-names></name> <name><surname>Banerjee</surname> <given-names>M.</given-names></name> <name><surname>Vemuri</surname> <given-names>B. C.</given-names></name></person-group> (<year>2018b</year>). <article-title>H-CNNS: Convolutional neural networks for riemannian homogeneous spaces</article-title>. <source>arXiv [Preprint]. arXiv:1805.05487.05481</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1805.05487</pub-id></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chakraborty</surname> <given-names>R.</given-names></name> <name><surname>Bouza</surname> <given-names>J.</given-names></name> <name><surname>Manton</surname> <given-names>J.</given-names></name> <name><surname>Vemuri</surname> <given-names>B. C.</given-names></name></person-group> (<year>2020</year>). <article-title>Manifoldnet: A deep neural network for manifold-valued data with applications</article-title>. <source>IEEE Trans. Pattern Analy. Mach. Intellig</source>. <volume>44</volume>, <fpage>799</fpage>&#x02013;<lpage>810</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2020.3003846</pub-id><pub-id pub-id-type="pmid">32750791</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Chen</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Hill</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <source>Equivariant Point Network for 3D Point Cloud Analysis</source>, <fpage>14514</fpage>&#x02013;<lpage>14523</lpage>.<pub-id pub-id-type="pmid">36520758</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cohen</surname> <given-names>T.</given-names></name> <name><surname>Geiger</surname> <given-names>M.</given-names></name> <name><surname>K&#x000F6;hler</surname> <given-names>J.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Spherical CNNs,&#x0201D;</article-title> in <source>International Conference on Learning Representations</source> (<publisher-loc>Vancouver, BC</publisher-loc>).</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cohen</surname> <given-names>T.</given-names></name> <name><surname>Geiger</surname> <given-names>M.</given-names></name> <name><surname>Weller</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;A general theory of equivariant CNNs on homogeneous spaces,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems (NeurIPS 2019</source>), <fpage>9142</fpage>&#x02013;<lpage>9153</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cohen</surname> <given-names>T.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name></person-group> (<year>2016a</year>). <article-title>Group equivariant convolutional neural networks</article-title>. <source>Int. Conf. Mach. Learn</source>. <volume>2016</volume>, <fpage>2990</fpage>&#x02013;<lpage>2999</lpage>. <pub-id pub-id-type="doi">10.48550/arXiv.1602.07576</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cohen</surname> <given-names>T. S.</given-names></name> <name><surname>Weiler</surname> <given-names>M</given-names></name> <name><surname>Kicanaoglu</surname> <given-names>B.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Gauge equivariant convolutional networks and the icosahedral CNN,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Long Beach, CA</publisher-loc>), <fpage>1321</fpage>&#x02013;<lpage>1330</lpage>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cohen</surname> <given-names>T. S.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name></person-group> (<year>2016b</year>). <article-title>Steerable CNNs</article-title>. <source>arXiv</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1612.08498</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Diestel</surname> <given-names>J.</given-names></name> <name><surname>Spalsbury</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <source>The Joys of Haar Measure</source>. Amer Mathematical Society.</citation>
</ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Elaldi</surname> <given-names>A.</given-names></name> <name><surname>Dey</surname> <given-names>N.</given-names></name> <name><surname>Kim</surname> <given-names>H.</given-names></name> <name><surname>Gerig</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Equivariant spherical deconvolution: Learning sparse orientation distribution functions from spherical data,&#x0201D;</article-title> in <source>Information Processing in Medical Imaging</source>, eds. A. Feragen, S. Sommer, J. Schnabel, and M. Nielsen (Cham: Springer International Publishing), <fpage>267</fpage>&#x02013;<lpage>278</lpage>.<pub-id pub-id-type="pmid">37576905</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gens</surname> <given-names>R.</given-names></name> <name><surname>Domingos</surname> <given-names>P.</given-names></name></person-group> (<year>2014</year>). <source>Deep Symmetry networks</source>. Vancouver: <volume>NIPS</volume>, <fpage>2537</fpage>&#x02013;<lpage>2545</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gerken</surname> <given-names>J. E.</given-names></name> <name><surname>Aronsson</surname> <given-names>J.</given-names></name> <name><surname>Carlsson</surname> <given-names>O.</given-names></name> <name><surname>Linander</surname> <given-names>H.</given-names></name> <name><surname>Ohlsson</surname> <given-names>F.</given-names></name> <name><surname>Petersson</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Geometric deep learning and equivariant neural networks</article-title>. <source>Artif. Intell. Rev</source>. <volume>56</volume>, <fpage>14605</fpage>&#x02013;<lpage>14662</lpage>. <pub-id pub-id-type="doi">10.1007/s10462-023-10502-7</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Golkov</surname> <given-names>V.</given-names></name> <name><surname>Dosovitskit</surname> <given-names>A.</given-names></name> <name><surname>Sperl</surname> <given-names>J. I.</given-names></name> <name><surname>Menzel</surname> <given-names>M. I.</given-names></name> <name><surname>Czisch</surname> <given-names>M.</given-names></name> <name><surname>S&#x000E4;rmann</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title><italic>q</italic>-space deep learning: twelve-fold shorter and model-free diffusion MRI scans</article-title>. <source>IEEE Trans. Med</source>. <volume>35</volume>, <fpage>1344</fpage>&#x02013;<lpage>1351</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2016.2551324</pub-id><pub-id pub-id-type="pmid">27071165</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Graham</surname> <given-names>S.</given-names></name> <name><surname>Epstein</surname> <given-names>D.</given-names></name> <name><surname>Rajpoot</surname> <given-names>N.</given-names></name></person-group> (<year>2020</year>). <article-title>Dense steerable filter cnns for exploiting rotational symmetry in histology images</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>39</volume>, <fpage>4124</fpage>&#x02013;<lpage>4136</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2020.3013246</pub-id><pub-id pub-id-type="pmid">32746153</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jupp</surname> <given-names>P. E.</given-names></name> <name><surname>Mardia</surname> <given-names>K. V.</given-names></name></person-group> (<year>1989</year>). <article-title>A unified view of the theory of directional statistics, 1975-1988</article-title>. <source>Int. Statist. Rev</source>. <volume>57</volume>, <fpage>261</fpage>&#x02013;<lpage>294</lpage>. <pub-id pub-id-type="doi">10.2307/1403799</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Knigge</surname> <given-names>D. M.</given-names></name> <name><surname>Romero</surname> <given-names>D. W.</given-names></name> <name><surname>Bekkers</surname> <given-names>E. J.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Exploiting redundancy: separable group convolutional networks on lie groups,&#x0201D;</article-title> in <source>Proceedings of the 39th International Conference on Machine Learning</source>, eds. K. Chaudhuri, S. Jegelka, L. Song, C. Szepesvari, G. Niu, and S. Sabato (New York: PMLR), 11359-11386.</citation>
</ref>
<ref id="B27">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kondor</surname> <given-names>R.</given-names></name> <name><surname>Trivedi</surname> <given-names>S.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;On the generalization of equivariance and convolution in neural networks to the action of compact groups,&#x0201D;</article-title> in <source>International Conference on Machine Learning</source> (<publisher-loc>Stockholm</publisher-loc>), <fpage>2747</fpage>&#x02013;<lpage>2755</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>T.-Y.</given-names></name> <name><surname>Goyal</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <article-title>Focal loss for dense object detection</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>42</volume>, <fpage>318</fpage>&#x02013;<lpage>327</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2018.2858826</pub-id><pub-id pub-id-type="pmid">30040631</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>R.</given-names></name> <name><surname>Lauze</surname> <given-names>F.</given-names></name> <name><surname>Erleben</surname> <given-names>K.</given-names></name> <name><surname>Darkner</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Bundle geodesic convolutional neural network for dwi segmentation from single scan learning,&#x0201D;</article-title> in <source>Computational Diffusion MRI</source>, eds. S. Cetin-Karayumak, D. Christiaens, M. Figini, P. Guevara, N. Gyori, V. Nath, and T. Pieciak (Cham: Springer International Publishing), <fpage>121</fpage>&#x02013;<lpage>132</lpage>.</citation>
</ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>R.</given-names></name> <name><surname>Lauze</surname> <given-names>F. B.</given-names></name> <name><surname>Bekkers</surname> <given-names>E. J.</given-names></name> <name><surname>Erleben</surname> <given-names>K.</given-names></name> <name><surname>Darkner</surname> <given-names>S.</given-names></name></person-group> (<year>2022</year>). <article-title>&#x0201C;Group convolutional neural networks for DWI segmentation,&#x0201D;</article-title> in <source>Geometric Deep Learning in Medical Image Analysis</source> (<publisher-loc>Strasbourg</publisher-loc>).</citation>
</ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Masci</surname> <given-names>J.</given-names></name> <name><surname>Boscaini</surname> <given-names>D.</given-names></name> <name><surname>Bronstein</surname> <given-names>M.</given-names></name> <name><surname>Vandergheynst</surname> <given-names>P.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Geodesic convolutional neural networks on riemannian manifolds,&#x0201D;</article-title> in <source>2015 IEEE International Conference on Computer Vision Workshop (ICCVW)</source> (<publisher-loc>Santiago</publisher-loc>), <fpage>832</fpage>&#x02013;<lpage>840</lpage>. <pub-id pub-id-type="doi">10.1109/ICCVW.2015.112</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x000FC;ller</surname> <given-names>P.</given-names></name> <name><surname>Golkov</surname> <given-names>V.</given-names></name> <name><surname>Tomassini</surname> <given-names>V.</given-names></name> <name><surname>Cremers</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <source>Rotation-Equivariant Deep Learning for Diffusion MRI</source>. ISMRM.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Novikov</surname> <given-names>D.</given-names></name> <name><surname>Veraart</surname> <given-names>J.</given-names></name> <name><surname>Jelescu</surname> <given-names>I.</given-names></name> <name><surname>Fieremans</surname> <given-names>E.</given-names></name></person-group> (<year>2018</year>). <article-title>Rotationally-invariant mapping of scalar and orientational metrics of neuronal microstructure with diffusion MRI</article-title>. <source>Neuroimage</source> <volume>174</volume>, <fpage>518</fpage>&#x02013;<lpage>538</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2018.03.006</pub-id><pub-id pub-id-type="pmid">29544816</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Poulenard</surname> <given-names>A.</given-names></name> <name><surname>Ovsjanikov</surname> <given-names>M.</given-names></name> <name><surname>Guibas</surname> <given-names>L. J.</given-names></name></person-group> (<year>2022</year>). <article-title>Equivalence between Se(3) equivariant networks via steerable kernels and group convolution</article-title>. <source>arXiv [Preprint] arXiv:2211.15903</source>.</citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schnell</surname> <given-names>S.</given-names></name> <name><surname>Saur</surname> <given-names>D.</given-names></name> <name><surname>Kreher</surname> <given-names>B. W.</given-names></name> <name><surname>Hennig</surname> <given-names>J.</given-names></name> <name><surname>Burkhardt</surname> <given-names>H.</given-names></name> <name><surname>Kiselev</surname> <given-names>V. G.</given-names></name></person-group> (<year>2009</year>). <article-title>Fully automated classification of hardi in vivo data using a support vector machine</article-title>. <source>Neuroimage</source> <volume>46</volume>:<fpage>642</fpage>&#x02013;<lpage>651</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2009.03.003</pub-id><pub-id pub-id-type="pmid">19285561</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schonsheck</surname> <given-names>S. C.</given-names></name> <name><surname>Dong</surname> <given-names>B.</given-names></name> <name><surname>Lai</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>Parallel transport convolution: a new tool for convolutional neural networks on manifolds</article-title>. <source>arXiv [Preprint] arXiv:1805.07857</source>.</citation>
</ref>
<ref id="B37">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Schwab</surname> <given-names>E.</given-names></name> <name><surname>Ceting&#x000FC;l</surname> <given-names>H. E.</given-names></name> <name><surname>Asfari</surname> <given-names>B.</given-names></name> <name><surname>Vidal</surname> <given-names>E.</given-names></name></person-group> (<year>2013</year>). <article-title>&#x0201C;Rotational invariant features for HARDI,&#x0201D;</article-title> in <source>Information Processing in Medical Imaging. Lecture Notes in Computer Science, Vol. 7917</source> (<publisher-loc>Berlin; Heidelberg</publisher-loc>: <publisher-name>Springer</publisher-name>).</citation>
</ref>
<ref id="B38">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sedlar</surname> <given-names>S.</given-names></name> <name><surname>Alimi</surname> <given-names>A.</given-names></name> <name><surname>Papadopoulo</surname> <given-names>T.</given-names></name> <name><surname>Deriche</surname> <given-names>R.</given-names></name> <name><surname>Deslauriers-Gauthier</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;A spherical convolutional neural network for white matter structure imaging via dMRI,&#x0201D;</article-title> in <source>Medical Image Computing and Computer Assisted Intervention-MICCAI 2021</source>, eds. M. de Bruijne, P. C. Cattin, S. Cotin, N. Padoy, S. Speidel, Y., Zheng, and C. Essert (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>529</fpage>&#x02013;<lpage>539</lpage>.</citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sedlar</surname> <given-names>S.</given-names></name> <name><surname>Papadopoulo</surname> <given-names>T.</given-names></name> <name><surname>Deriche</surname> <given-names>R.</given-names></name> <name><surname>Deslauriers-Gauthier</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Diffusion MRI fiber orientation distribution function estimation using voxel-wise spherical U-net,&#x0201D;</article-title> in <source>International MICCAI Workshop 2020</source> - <italic>Computational Diffusion MRI</italic> (Lima: MICCAI).</citation>
</ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Skibbe</surname> <given-names>H.</given-names></name> <name><surname>Reisert</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). <article-title>Spherical tensor algebra: a toolkit for 3d image processing</article-title>. <source>J. Math. Imaging Vis</source>. <volume>58</volume>, <fpage>349</fpage>&#x02013;<lpage>381</lpage>. <pub-id pub-id-type="doi">10.1007/s10851-017-0715-7</pub-id></citation>
</ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Smets</surname> <given-names>B. M. N.</given-names></name> <name><surname>Portegies</surname> <given-names>J.</given-names></name> <name><surname>Bekkers</surname> <given-names>E. J.</given-names></name> <name><surname>Duits</surname> <given-names>R.</given-names></name></person-group> (<year>2021</year>). <source>PDE-based Group Equivariant Convolutional Neural Networks</source>. <publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sommer</surname> <given-names>S.</given-names></name> <name><surname>Bronstein</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>Horizontal flows and manifold stochastics in geometric deep learning</article-title>. <source>IEEE Trans. PAMI</source>. <volume>44</volume>, <fpage>811</fpage>&#x02013;<lpage>822</lpage>. <pub-id pub-id-type="doi">10.1109/TPAMI.2020.2994507</pub-id><pub-id pub-id-type="pmid">32406826</pub-id></citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tuchs</surname> <given-names>D. S.</given-names></name></person-group> (<year>2004</year>). <article-title>Q-ball imaging</article-title>. <source>Magnet. Reson. Med</source>. <volume>52</volume>, <fpage>1358</fpage>&#x02013;<lpage>1372</lpage>. <pub-id pub-id-type="doi">10.1002/mrm.20279</pub-id><pub-id pub-id-type="pmid">15562495</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van Essen</surname> <given-names>D. C.</given-names></name> <name><surname>Smith</surname> <given-names>S. M.</given-names></name> <name><surname>Barch</surname> <given-names>D. M.</given-names></name> <name><surname>Behrens</surname> <given-names>T. E.</given-names></name> <name><surname>Yacoub</surname> <given-names>E.</given-names></name> <name><surname>Ugurbil</surname> <given-names>K.</given-names></name></person-group> (<year>2013</year>). <article-title>The WU-Minn human connectome project: an overview</article-title>. <source>Neuroimage</source> <volume>80</volume>, <fpage>62</fpage>&#x02013;<lpage>79</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2013.05.041</pub-id><pub-id pub-id-type="pmid">23684880</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weiler</surname> <given-names>M.</given-names></name> <name><surname>Forr&#x000E9;</surname> <given-names>P.</given-names></name> <name><surname>Verlinde</surname> <given-names>E.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Convolutional networks-isometry and gauge equivariant convolutions on riemannian manifolds</article-title>. <source>arXiv [Preprint] arXiv:2106.06020</source>. <pub-id pub-id-type="doi">10.48550/arXiv.2106.06020</pub-id></citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weiler</surname> <given-names>M.</given-names></name> <name><surname>Geiger</surname> <given-names>M.</given-names></name> <name><surname>Welling</surname> <given-names>M.</given-names></name> <name><surname>Boomsma</surname> <given-names>W.</given-names></name> <name><surname>Cohen</surname> <given-names>T.</given-names></name></person-group> (<year>2018a</year>). <article-title>&#x0201C;3D steerable CNNs: learning rotationally equivariant features in volumetric data,&#x0201D;</article-title> in <source>Advances in Neural Information Processing Systems</source>, <fpage>10401</fpage>&#x02013;<lpage>10412</lpage>.</citation>
</ref>
<ref id="B47">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Weiler</surname> <given-names>M.</given-names></name> <name><surname>Hamprecht</surname> <given-names>F.</given-names></name> <name><surname>Storath</surname> <given-names>M.</given-names></name></person-group> (<year>2018b</year>). <article-title>&#x0201C;Learning steerable filters for rotation equivariant CNNS,&#x0201D;</article-title> in <source>2018 IEEE/CVF Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>849</fpage>&#x02013;<lpage>858</lpage>.</citation>
</ref>
<ref id="B48">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Worrall</surname> <given-names>D.</given-names></name> <name><surname>Garbin</surname> <given-names>S.</given-names></name> <name><surname>Turmukhambetov</surname> <given-names>D.</given-names></name> <name><surname>Brostow</surname> <given-names>G.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Harmonic networks: deep translation and rotation equivariance,&#x0201D;</article-title> in <source>2017 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>).</citation>
</ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zucchelli</surname> <given-names>M.</given-names></name> <name><surname>Deslauriers-Gauthier</surname> <given-names>S.</given-names></name> <name><surname>Deriche</surname> <given-names>R.</given-names></name></person-group> (<year>2020</year>). <article-title>A computational framework for generating rotation invariant features and its application in diffusion MRI</article-title>. <source>Med. Image Anal</source>. <volume>60</volume>:<fpage>101597</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2019.101597</pub-id><pub-id pub-id-type="pmid">31810004</pub-id></citation></ref>
</ref-list>
</back>
</article>