<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2025.1618426</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Thyroid nodule segmentation in ultrasound images using transformer models with masked autoencoder pre-training</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Xiang</surname> <given-names>Yi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2894880/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Acharya</surname> <given-names>Rajendra</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Le</surname> <given-names>Quan</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Tan</surname> <given-names>Jen Hong</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1170808/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Chng</surname> <given-names>Chiaw-Ling</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/496292/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Office of Insights &#x00026; Analytics, Division of Digital Strategy, SingHealth</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country></aff>
<aff id="aff2"><sup>2</sup><institution>School of Mathematics, Physics and Computing, University of Southern Queensland</institution>, <addr-line>Springfield Central, QLD</addr-line>, <country>Australia</country></aff>
<aff id="aff3"><sup>3</sup><institution>Data Science and Artificial Intelligence Lab, Singapore General Hospital</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Endocrinology, Singapore General Hospital</institution>, <addr-line>Singapore</addr-line>, <country>Singapore</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Norberto Peporine Lopes, University of S&#x000E3;o Paulo, Bauru, Brazil</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Amir Faisal, Sumatra Institute of Technology, Indonesia</p>
<p>Guohui Wei, Shandong University of Traditional Chinese Medicine, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Yi Xiang <email>xiangyi&#x00040;u.nus.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>24</day>
<month>07</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>8</volume>
<elocation-id>1618426</elocation-id>
<history>
<date date-type="received">
<day>26</day>
<month>04</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>07</day>
<month>07</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2025 Xiang, Acharya, Le, Tan and Chng.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Xiang, Acharya, Le, Tan and Chng</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Thyroid nodule segmentation in ultrasound (US) images is a valuable yet challenging task, playing a critical role in diagnosing thyroid cancer. The difficulty arises from factors such as the absence of prior knowledge about the thyroid region, low contrast between anatomical structures, and speckle noise, all of which obscure boundary detection and introduce variability in nodule appearance across different images.</p></sec>
<sec>
<title>Methods</title>
<p>To address these challenges, we propose a transformer-based model for thyroid nodule segmentation. Unlike traditional convolutional neural networks (CNNs), transformers capture global context from the first layer, enabling more comprehensive image representation, which is crucial for identifying subtle nodule boundaries. In this study, We first pre-train a Masked Autoencoder (MAE) to reconstruct masked patches, then fine-tune on thyroid US data, and further explore a cross-attention mechanism to enhance information flow between encoder and decoder.</p></sec>
<sec>
<title>Results</title>
<p>Our experiments on the public AIMI, TN3K, and DDTI datasets show that MAE pre-training accelerates convergence. However, overall improvements are modest: the model achieves Dice Similarity Coefficient (DSC) scores of 0.63, 0.64, and 0.65 on AIMI, TN3K, and DDTI, respectively, highlighting limitations under small-sample conditions. Furthermore, adding cross-attention did not yield consistent gains, suggesting that data volume and diversity may be more critical than additional architectural complexity.</p></sec>
<sec>
<title>Discussion</title>
<p>MAE pre-training notably reduces training time and helps themodel learn transferable features, yet overall accuracy remains constrained by limited data and nodule variability. Future work will focus on scaling up data, pre-training cross-attention layers, and exploring hybrid architectures to further boost segmentation performance.</p></sec></abstract>
<kwd-group>
<kwd>thyroid nodule segmentation</kwd>
<kwd>ultrasound imaging</kwd>
<kwd>transformer-based network</kwd>
<kwd>Masked Autoencoder</kwd>
<kwd>self-supervised learning</kwd>
</kwd-group>
<contract-sponsor id="cn001">Singapore General Hospital<named-content content-type="fundref-id">https://doi.org/10.13039/501100001469</named-content></contract-sponsor>
<counts>
<fig-count count="7"/>
<table-count count="3"/>
<equation-count count="2"/>
<ref-count count="19"/>
<page-count count="10"/>
<word-count count="5288"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Medicine and Public Health</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>Thyroid nodules are commonly found in the general population and are often detected through imaging, either during investigations for thyroid-related issues or as incidental findings (Dean and Gharib, <xref ref-type="bibr" rid="B3">2008</xref>). While most nodules are benign and asymptomatic, a small percentage can be malignant, requiring timely and accurate evaluation. Segmentation is an essential initial step in this process, as it delineates the interface between the nodule and the surrounding parenchyma, aiding in the assessment of malignancy likelihood.</p>
<p>Segmenting thyroid nodules in ultrasound images presents several technical challenges. The lack of distinct anatomical landmarks, coupled with the low contrast between tissues, makes it difficult to differentiate the boundaries of the nodules. Additionally, the granular speckle noise inherent in ultrasound imaging adds further complexity by distorting image clarity and increasing the variability of nodule shapes and appearances across frames.</p>
<p>Early segmentation methods, such as K-means (Hart et al., <xref ref-type="bibr" rid="B7">2000</xref>), Fuzzy C-means (FCM) (Hart et al., <xref ref-type="bibr" rid="B7">2000</xref>), efficient graph-based segmentation (EGB) (Felzenszwalb and Huttenlocher, <xref ref-type="bibr" rid="B5">2004</xref>), and robust graph-based segmentation (RGB) (Huang et al., <xref ref-type="bibr" rid="B9">2012</xref>), were among the earliest techniques applied in computer-aided image segmentation. These methods rely on pre-defined parameters, such as thresholds, which are manually crafted and need careful tuning for optimal performance (Xu et al., <xref ref-type="bibr" rid="B19">2019</xref>). Despite their simplicity and effectiveness in certain cases, these methods often struggle to adapt to complex patterns in large and diverse datasets.</p>
<p>With the advancement of data availability and computational power, deep learning-based approaches, particularly convolutional neural networks (CNNs), have gained prominence. CNNs excel at capturing patterns between inputs and outputs by learning features directly from data, eliminating the need for handcrafted features. Traditional CNNs for segmentation often utilize an encoder-decoder architecture. The encoder extracts low-resolution feature maps, while the decoder up-samples these maps to produce per-pixel class predictions. Fully Convolutional Networks (FCNs) (Long et al., <xref ref-type="bibr" rid="B10">2015</xref>) represent a classic implementation of this architecture. Building on this, models such as U-Net (Ronneberger et al., <xref ref-type="bibr" rid="B16">2015</xref>) introduced skip connections between the encoder and decoder, allowing the combination of fine-grained details from earlier layers with deeper, more abstract features, thus preserving critical spatial information.</p>
<p>However, one key limitation of CNNs is their inability to effectively capture global image context due to the local nature of convolutional filters. Although deeper layers in CNNs can expand the receptive field, they still struggle to form a comprehensive view of the entire image, making it difficult to accurately segment structures like thyroid nodules. Transformers (Vaswani, <xref ref-type="bibr" rid="B18">2017</xref>), in contrast, inherently capture global context from the very first layer, offering a more holistic image representation that is crucial for detecting subtle boundary variations. However, despite their strengths, Vision Transformers (ViTs) tend to underperform on smaller datasets compared to CNNs, as they lack the inductive biases that help CNNs generalize well with limited data (Dosovitskiy, <xref ref-type="bibr" rid="B4">2020</xref>).</p>
<p>In this study, we deviate from the typical approach of pre-training ViTs on classification datasets like ImageNet-21k. Instead, we aim to leverage the segmentation dataset more effectively by pre-training the model using a Masked Autoencoder (MAE) (He et al., <xref ref-type="bibr" rid="B8">2022</xref>). The MAE reconstructs partially masked images, enabling the model to capture image patterns more effectively for segmentation tasks.</p>
<p>We perform an extensive analysis of transformer architectures for segmentation, experimenting with different model architectures and input patch sizes. Inspired by advancements in natural language processing (Vaswani, <xref ref-type="bibr" rid="B18">2017</xref>), we incorporate a cross-attention mechanism to improve the interaction between the encoder and decoder, enhancing the model&#x00027;s context capture capabilities.</p>
<p>In summary, we propose a transformer-based approach with MAE pre-training for thyroid nodule segmentation, performing ablation studies on model architectures and patch sizes to optimize performance for this challenging task.</p></sec>
<sec sec-type="methods" id="s2">
<title>2 Methods</title>
<sec>
<title>2.1 Dataset</title>
<p>For the pre-training and segmentation tasks, we utilize three open-source datasets: AIMI, TN3K (Gong et al., <xref ref-type="bibr" rid="B6">2021</xref>), and DDTI (Pedraza et al., <xref ref-type="bibr" rid="B12">2015</xref>). The AIMI dataset was collected from 167 patients with 192 biopsy-confirmed thyroid nodules at the Stanford University Medical Center. The TN3K dataset consists of ultrasound images provided in Gong et al. (<xref ref-type="bibr" rid="B6">2021</xref>), while the DDTI dataset was compiled with the support of the Universidad Nacional de Colombia. The specifics of each dataset, including the number of images and the corresponding patient data, are outlined in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Dataset and splitting details.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Total number of images</bold></th>
<th valign="top" align="center"><bold>Training images</bold></th>
<th valign="top" align="center"><bold>Testing images</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AIMI</td>
<td valign="top" align="center">17,412 (from 192 subjects)</td>
<td valign="top" align="center">14,055 (from 154 subjects)</td>
<td valign="top" align="center">3,357 (from 38 subjects)</td>
</tr> <tr>
<td valign="top" align="left">TN3K</td>
<td valign="top" align="center">3,493</td>
<td valign="top" align="center">2,879</td>
<td valign="top" align="center">614</td>
</tr> <tr>
<td valign="top" align="left">DDTI</td>
<td valign="top" align="center">637</td>
<td valign="top" align="center">477</td>
<td valign="top" align="center">160</td>
</tr></tbody>
</table>
</table-wrap>
<p>For model training and evaluation, we split each dataset into training and testing subsets. The AIMI and TN3K datasets were split in an 80:20 ratio, while the DDTI dataset was split in a 75:25 ratio. The exact number of images used for training and testing across each dataset is presented in <xref ref-type="table" rid="T1">Table 1</xref>.</p>
<p>We applied these splits in two key stages of the process:</p>
<list list-type="bullet">
<list-item><p>For MAE pre-training, we used only the training portion of each dataset to pre-train the model.</p></list-item>
<list-item><p>For the segmentation model, we trained the model on the training subset and evaluated its performance on the testing subset to assess generalization capabilities.</p></list-item>
</list></sec>
<sec>
<title>2.2 Masked autoencoder (MAE)</title>
<p>We follow the self-supervised training framework outlined in the original MAE paper (He et al., <xref ref-type="bibr" rid="B8">2022</xref>), utilizing an encoder-decoder architecture to pre-train a Vision Transformer (ViT). In this process, the input ultrasound images are first resize to 224 &#x000D7; 224 and then divided into non-overlapping patches of size 14 &#x000D7; 14. We randomly mask 75% of the patches, as recommended in original MAE paper.</p>
<p>The remaining 25% of the unmasked patches are fed into the encoder, a 12-layer ViT with a patch embedding size of 192 and three attention heads. The masked patches are not passed through the encoder but are later included in the decoder stage. The decoder, consisting of a four-layer ViT with three attention heads, receives both the encoded unmasked patches and a learned representation for the masked patches. The decoder then reconstructs the entire image, and a linear projection layer maps the output back to the original image resolution.</p>
<p>In the original MAE framework, reconstruction loss is measured using mean squared error (MSE) on the masked patches. We extend this by incorporating loss from the unmasked patches as well to enhance the model&#x00027;s attention to both masked and visible regions. Our modified loss function is:</p>
<disp-formula id="E1"><mml:math id="M1"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mtext>_</mml:mtext><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mi>&#x003B1;</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>u</mml:mi><mml:mi>n</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>k</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mtext>_</mml:mtext><mml:mi>p</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>c</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where &#x003B1; is a weighting factor that balances the contribution of the unmasked patches to the total loss. We empirically set &#x003B1; &#x0003D; 0.1 to encourage focus on the reconstruction of masked regions while still accounting for some information from unmasked patches.</p>
<p>We train the MAE using the AdamW optimizer with an initial learning rate of 2 &#x000D7; 10<sup>&#x02212;5</sup>, a batch size of 1.024, and 4.000 epochs. Data augmentation techniques, including random horizontal flips and random resized crops, are applied to increase the diversity of training data and improve the model&#x00027;s generalizability, especially given the relatively small size of the thyroid nodule dataset.</p></sec>
<sec>
<title>2.3 Segmentation model architecture</title>
<p>The segmentation model largely follows the same structure as the MAE pre-training process, with key differences in how all image patches are processed. After resizing and dividing the input ultrasound images into non-overlapping patches, all patches are passed through the encoder and decoder layers. We maintain the similar architecture as in the MAE process, using a 12-layer encoder and a three-layer decoder, to facilitate the comparison between the model trained from scratch and the one fine-tuned with MAE pre-trained weights.</p>
<p>Specifically, the input images are resized to (224 &#x000D7; 224) pixels and divided into (14 &#x000D7; 14) patches, resulting in 256 patches per image. After flattening each patch, a linear layer projects the resulting vectors into a 192-dimensional space. Positional embeddings are then added, resulting in an input tensor of shape (16 &#x000D7; 16, 192). This tensor is passed through the 12-layer encoder followed by the three-layer decoder. The final output of the decoder is projected to 196 dimensions (corresponding to the flattened segmentation map) and then passed through a sigmoid activation function to generate the segmentation mask, as illustrated in <xref ref-type="fig" rid="F1">Figure 1</xref>. The total number of parameters in this configuration is 4.66M.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The architecture of the segmentation model. The division of patches at here is for illustration. Please refer to section 2.3 for actual setup.</p></caption>
<alt-text>Flowchart of an image processing model. The process starts with a 224x224 image, divided into patches. Each patch undergoes linear transformation, positional embedding, and is fed into encoder layers one to twelve, each having a 16x16x192 dimension. Subsequent decoder layers further process, culminating in a linear transformation for patch reassembly. The output shows segmented patches reassembled into a final image</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1618426-g0001.tif"/>
</fig>
<p>We train the segmentation model for 400 epochs using an initial learning rate of 1.6 &#x000D7; 10<sup>&#x02212;4</sup> and a batch size of 512. The Dice Loss is employed as the loss function to directly optimize for segmentation performance. Data augmentation techniques, including random rotation, random horizontal flips, and random resized cropping, are applied to enhance model generalization, especially given the limited size of the training dataset.</p></sec>
<sec>
<title>2.4 Cross-attention architecture</title>
<p>In addition to the traditional encoder-decoder architecture using Vision Transformer (ViT) layers, we also explore a cross-attention mechanism within the segmentation model. Cross-attention, originally introduced in natural language processing (NLP) (Vaswani, <xref ref-type="bibr" rid="B18">2017</xref>), connects the encoder and decoder by allowing information exchange between the two, rather than relying solely on self-attention. This enhances the model&#x00027;s ability to leverage feature representations at multiple levels, similar to U-Net&#x00027;s skip connections.</p>
<p>The architecture we implemented is illustrated in <xref ref-type="fig" rid="F2">Figure 2</xref>. In each cross-attention layer, the attention mechanism is computed as:</p>
<disp-formula id="E2"><label>(1)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>t</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>s</mml:mi><mml:mi>o</mml:mi><mml:mi>f</mml:mi><mml:mi>t</mml:mi><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:mi>&#x00040;</mml:mi><mml:msup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mi>V</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In this case, the query (<italic>Q</italic>) comes from the decoder, while the key (<italic>K</italic>) and value (<italic>V</italic>) are sourced from the encoder output. This differs from the self-attention mechanism, where the query, key, and value all originate from the same source (either the encoder or decoder). Cross-attention allows the model to focus on relevant regions in the encoder output while processing the decoder&#x00027;s output, effectively linking the two stages. After incorporating the cross-attention layer, the total number of parameters increases to 5.54M, representing a 18.88% increase compared to the architecture without cross-attention.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The architect of the segmentation model with cross attention: connecting the encoder and decoder similar to a U-net network. The division of patches at here is for illustration. Please refer to section 2.4 for actual setup.</p></caption>
<alt-text>Flowchart of a neural network process. It shows how an image is divided into patches and processed through layers. The patches undergo linear transformations, positional embedding, cross attention, and encoding-decoding through twelve encoder layers and three decoder layers. It illustrates integration of cross attention for feature extraction and final patch reassembly.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1618426-g0002.tif"/>
</fig>
<p>This type of cross-attention architecture is widely used in NLP, most notably in models like T5 (Raffel et al., <xref ref-type="bibr" rid="B15">2020</xref>), which employs an encoder-decoder structure with cross-attention to enhance information flow between the two components.</p>
<p>For our segmentation task, we designed a cross-attention-based architecture similar to U-Net, adding skip connections from the encoder to the decoder. This modification aims to better preserve spatial details and contextual information during the decoding process. We compare the performance of this cross-attention architecture with the original segmentation architecture in the following experiments to assess the impact of this design on segmentation accuracy and boundary detection.</p></sec></sec>
<sec sec-type="results" id="s3">
<title>3 Results</title>
<sec>
<title>3.1 Masked autoencoder (MAE)</title>
<p>We pre-trained the MAE model using the training dataset (size = 17,411) over 4,000 epochs. To evaluate the influence of patch size, we experimented with two configurations: a smaller patch size of 9 &#x000D7; 9 and a larger patch size of 14 &#x000D7; 14. The loss curves for both configurations are shown in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>MAE loss curves for different patch sizes over 4,000 epochs. The pink curve represents the loss for patch size 9 &#x000D7; 9, and the red curve represents the loss for patch size 14 &#x000D7; 14.</p></caption>
<alt-text>Line graph showing loss versus epoch for two different patch sizes. The red line represents patch size fourteen by fourteen, and the pink line represents patch size nine by nine. Both lines show a decreasing trend, with the pink line consistently lower, indicating lower loss for smaller patch sizes. The x-axis is labeled as Epoch, and the y-axis is labeled as Loss.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1618426-g0003.tif"/>
</fig>
<p>At the end of training (epoch 4,000), the MAE with a patch size of 9 &#x000D7; 9 achieved a lower reconstruction loss of 0.01698 compared to 0.02311 for the patch size of 14 &#x000D7; 14. As illustrated by the graph, the loss for patch size 9 decreased more rapidly during the initial training stages and outperformed patch size 14 throughout the training process, indicating that smaller patches facilitate better reconstruction performance.</p>
<p>We further evaluated the quality of the reconstructed images at various training stages: epochs 200, 2,000, and 4,000. <xref ref-type="fig" rid="F4">Figure 4</xref> demonstrate the evolution of the reconstructed outputs for both patch sizes. The results show that as training progresses, the quality of the reconstructions improves significantly for both patch sizes.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Reconstructed outputs by MAE with patch sizes 9 &#x000D7; 9 and 14 &#x000D7; 14 at different training stages (epochs 200, 2000, 4000). For each triplet, the original image is on the left, the masked input is in the middle, and the MAE-reconstructed image is on the right.</p></caption>
<alt-text>Comparison of image reconstructions at different epochs with patch sizes. The left column shows a 9x9 patch size and the right column a 14x14 patch size. Each set includes ultrasound images at epochs 200, 2000, and 4000. Images show varying levels of noise and reconstruction detail based on patch size and epochs.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1618426-g0004.tif"/>
</fig>
<p><xref ref-type="fig" rid="F4">Figure 4</xref> also compares the reconstruction performance of the two patch sizes. The left column displays results for patch size 9 &#x000D7; 9, while the right column corresponds to patch size 14 &#x000D7; 14. Notably, the 9 &#x000D7; 9 patch size produced more detailed reconstructions, successfully capturing finer image features compared to the 14 &#x000D7; 14 patch size, which led to slightly coarser results. This suggests that smaller patch sizes enable the model to learn and preserve more intricate details during the reconstruction process.</p></sec>
<sec>
<title>3.2 Segmentation models</title>
<sec>
<title>3.2.1 Model performance pretrained with MAE vs. without</title>
<p>Given the architectural similarity between the encoder-decoder structure in MAE and the segmentation model, we trained the segmentation model both from scratch and using weights pretrained from the MAE process. Since the MAE decoder has four layers, while the segmentation model&#x00027;s decoder consists of three layers, we dropped the last layer of the MAE model when loading the weights.</p>
<p>We set the total training epochs to 400. As shown in <xref ref-type="fig" rid="F5">Figure 5</xref>, the loss curve for training from scratch decreases slowly, leading us to apply early stopping. Subsequent experiments use the segmentation model initialized with the pretrained MAE weights.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Segmentation loss curve for patch size 14. The blue line represents the training process from scratch, the green line represents training using weights pretrained in MAE without the cross-attention architecture, and the purple line represents training with cross-attention architecture.</p></caption>
<alt-text>Line graph showing loss versus epoch for three training methods. Blue line: training from scratch, maintaining higher loss. Green line: pretraining without cross-attention, reducing loss. Pink line: pretraining with cross-attention, achieving the lowest loss.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1618426-g0005.tif"/>
</fig>
</sec>
<sec>
<title>3.2.2 Performance with different architectures and patch sizes</title>
<p>As demonstrated in the previous section, the pretraining process significantly improved performance and reduced training time. We further compared different model architectures and patch sizes after pretraining.</p>
<p>The Dice score performance of our transformer-based segmentation model is shown in <xref ref-type="table" rid="T2">Table 2</xref>, alongside results from baseline counterparts, including UNet (Ronneberger et al., <xref ref-type="bibr" rid="B16">2015</xref>), Attention UNet (Oktay et al., <xref ref-type="bibr" rid="B11">2018</xref>), SResUNet-AD (Radhachandran et al., <xref ref-type="bibr" rid="B14">2024</xref>), BPAT-UNet (Bi et al., <xref ref-type="bibr" rid="B1">2023</xref>), UNet Transformer (Petit et al., <xref ref-type="bibr" rid="B13">2021</xref>), and TransUNet (Chen et al., <xref ref-type="bibr" rid="B2">2021</xref>). While the data splitting methods for these baselines are not entirely identical, the results remain comparable. Our model demonstrates notable improvements over SResUNet-AD, which primarily excels at reducing false positives; this advantage may be less relevant in the AIMI and DDTI datasets, as they exclusively include nodule images. However, our model&#x00027;s transformer-based architecture, without any convolutional neural network (CNN) layers, may explain its inferior performance compared to CNN-based models or hybrid approaches that integrate CNN and transformer architectures.</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Dice score comparison of segmentation models across datasets and patch sizes.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center" colspan="3"><bold>Dice score</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>AIMI</bold></th>
<th valign="top" align="center"><bold>TN3K</bold></th>
<th valign="top" align="center"><bold>DDTI</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">UNet (Ronneberger et al., <xref ref-type="bibr" rid="B16">2015</xref>)</td>
<td valign="top" align="center">0.7003</td>
<td valign="top" align="center">0.7998</td>
<td valign="top" align="center">0.6983</td>
</tr> <tr>
<td valign="top" align="left">Attention UNet (Oktay et al., <xref ref-type="bibr" rid="B11">2018</xref>)</td>
<td valign="top" align="center"><bold>0.7129</bold></td>
<td valign="top" align="center">0.8114</td>
<td valign="top" align="center">0.7105</td>
</tr> <tr>
<td valign="top" align="left">SResUNet-AD (Radhachandran et al., <xref ref-type="bibr" rid="B14">2024</xref>)</td>
<td valign="top" align="center">0.5920 &#x000B1; 0.369</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.4020 &#x000B1; 0.384</td>
</tr> <tr>
<td valign="top" align="left">BPAT-UNet (Bi et al., <xref ref-type="bibr" rid="B1">2023</xref>)</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center"><bold>0.8364</bold></td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">Unet Transformer (Petit et al., <xref ref-type="bibr" rid="B13">2021</xref>)</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.8080</td>
<td valign="top" align="center">&#x02013;</td>
</tr> <tr>
<td valign="top" align="left">TransUNet (Chen et al., <xref ref-type="bibr" rid="B2">2021</xref>)</td>
<td valign="top" align="center">&#x02013;</td>
<td valign="top" align="center">0.8098</td>
<td valign="top" align="center"><bold>0.8350</bold></td>
</tr> <tr>
<td valign="top" align="left">With-cross attention-9 &#x000D7; 9</td>
<td valign="top" align="center">0.6254</td>
<td valign="top" align="center">0.6173</td>
<td valign="top" align="center">0.6537</td>
</tr> <tr>
<td valign="top" align="left">Without-cross attention-9 &#x000D7; 9</td>
<td valign="top" align="center">0.6182</td>
<td valign="top" align="center">0.6342</td>
<td valign="top" align="center">0.6555</td>
</tr> <tr>
<td valign="top" align="left">With-cross attention-14 &#x000D7; 14</td>
<td valign="top" align="center">0.6304</td>
<td valign="top" align="center">0.6390</td>
<td valign="top" align="center">0.6653</td>
</tr> <tr>
<td valign="top" align="left">Without-cross attention-14 &#x000D7; 14</td>
<td valign="top" align="center">0.6321</td>
<td valign="top" align="center">0.6354</td>
<td valign="top" align="center">0.6479</td>
</tr></tbody>
</table>
</table-wrap>
<p>We evaluated two patch sizes, 9 &#x000D7; 9 and 14 &#x000D7; 14, and two architectures: one with cross-attention (<xref ref-type="fig" rid="F2">Figure 2</xref>) and one without cross-attention (<xref ref-type="fig" rid="F1">Figure 1</xref>). The Dice scores for different datasets are reported in <xref ref-type="table" rid="T2">Table 2</xref>. Overall, the performance difference between models with and without cross-attention was not substantial, though there were some minor improvements in specific datasets.</p>
<p>Next, we examined the segmentation results. <xref ref-type="fig" rid="F6">Figures 6</xref>, <xref ref-type="fig" rid="F7">7</xref> show sample images from the training dataset, presenting segmentation results at epochs 100, 200, 300, and 400. Over the training process, the margins evolved from showing a significant mosaic effect to having smoother boundaries. While patch size 9 &#x000D7; 9 suffered from a more pronounced mosaic effect, it achieved higher positional accuracy compared to patch size 14 &#x000D7; 14. The difference in results between models with and without cross-attention was minor.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Segmentation results at different epochs (100, 200, 300, and 400) for patch size 9 &#x000D7; 9 with and without cross-attention. Each quadruplet includes the original ultrasound image, the gold standard, the predicted mask using the cross-attention model, and the predicted mask using the model without cross-attention.</p></caption>
<alt-text>Ultrasound image segmentation progress across four epochs (100, 200, 300, 400) with patch size nine by nine. Each row shows the original ultrasound image followed by three segmentation outputs, illustrating the development in segmentation accuracy and refinement over epochs.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1618426-g0006.tif"/>
</fig>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Segmentation results at different epochs (100, 200, 300, and 400) for patch size 14 &#x000D7; 14 with and without cross-attention. Each quadruplet includes the original ultrasound image, the gold standard, the predicted mask using the cross-attention model, and the predicted mask using the model without cross-attention.</p></caption>
<alt-text>Ultrasound images segmented across four training epochs: 100, 200, 300, and 400. Each row shows an original ultrasound scan followed by three stages of segmentation, illustrating progressive refinement and clarity in the segmented regions with increasing epochs. Patch size is fourteen by fourteen.</alt-text>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="frai-08-1618426-g0007.tif"/>
</fig>
</sec></sec></sec>
<sec sec-type="discussion" id="s4">
<title>4 Discussion</title>
<p>The results demonstrate that using Masked Autoencoder (MAE) pretraining significantly improves the efficiency of the segmentation model training process. By initializing the segmentation model with weights pre-trained on the MAE task, we were able to achieve faster convergence compared to training the segmentation model from scratch. This suggests that MAE effectively transfers useful features, allowing the segmentation model to fully utilize the available training data and reduce overall training time.</p>
<p>To further analyze the proposed method, we compare its advantages and limitations with those of traditional CNN-based approaches and hybrid CNN-Transformer architectures. <xref ref-type="table" rid="T3">Table 3</xref> summarizes the comparison.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Advantages and limitations of different methods.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="left"><bold>Advantages</bold></th>
<th valign="top" align="left"><bold>Limitations</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Traditional CNNs (e.g., U-Net, Attention U-Net, SResUNet-AD)</td>
<td valign="top" align="left">- Effectively captures complex patterns between inputs and outputs.<break/> - Well-established and widely used in medical imaging.</td>
<td valign="top" align="left">- Limited ability to capture global context due to local receptive fields.<break/> - Requires large-scale pretraining datasets (e.g., ImageNet).</td>
</tr> <tr>
<td valign="top" align="left">Hybrid CNN-Transformer Networks (e.g., BPAT-UNet, UNet transformer, TransUNet)</td>
<td valign="top" align="left">- Combines CNN&#x00027;s strength in extracting local features with Transformer&#x00027;s global context understanding.<break/> - High accuracy and strong performance metrics reported in various studies.</td>
<td valign="top" align="left">- Computationally expensive.<break/> - Often tuned for specific tasks and datasets, which may limit generalization for new applications.</td>
</tr> <tr>
<td valign="top" align="left">Pure transformer networks with masked autoencoder pretraining(this study)</td>
<td valign="top" align="left">- Possesses the ability to capture the global context.<break/> - Masked Autoencoder (MAE) pretraining improves training efficiency by reducing training time and fully utilizing the dataset.<break/> - Eliminates the need for external datasets for pretraining.</td>
<td valign="top" align="left">- Mosaic artifacts in results, particularly with small patch sizes.<break/> - Moderate segmentation accuracy and dice scores compared to hybrid methods.</td>
</tr></tbody>
</table>
</table-wrap>
<p>Traditional CNN-based methods, such as U-Net and Attention U-Net, excel at extracting local features and benefit from extensive pretraining on datasets like ImageNet. However, their limited receptive fields constrain their ability to capture global context, making them less effective for tasks requiring fine-grained boundary detection, such as thyroid nodule segmentation. Hybrid CNN-Transformer architectures, such as BPAT-UNet and TransUNet, leverage the strengths of both CNNs and Transformers, achieving high performance metrics. Nevertheless, these models are computationally expensive and often require careful tuning for specific datasets.</p>
<p>The proposed method, based on pure Transformer architecture with MAE pretraining, addresses some of these challenges by capturing global context and improving training efficiency. However, as shown in the results, the incorporation of the cross-attention mechanism did not lead to significant improvements. This could be due to the lack of pretraining for the cross-attention layers, which limits their effectiveness given the constraints of the training data. Future work could explore methods to pre-train these layers or leverage larger datasets to enhance their potential.</p>
<p>In terms of patch size, the smaller patch size (9 &#x000D7; 9) demonstrated better feature extraction during MAE pretraining, as evidenced by more detailed reconstructions. However, this did not translate into improved segmentation performance, as the segmentation results exhibited more pronounced mosaic effects. This could be attributed to the increased number of parameters associated with smaller patches, requiring more training epochs to fully optimize.</p>
<p>Finally, the overall dice scores were not as high as anticipated across all datasets. This could be attributed to the inherent difficulty of the thyroid nodule segmentation task, which presents challenges due to the variable shapes and indistinct boundaries of the nodules. Future experiments could explore different training strategies or architectures to further enhance performance.</p>
<p>Beyond the technical metrics, our MAE-pretrained segmentation pipeline can be developed into a fully automated workflow that significantly reduces manual delineation by radiographers and radiologists&#x02014;especially when processing large volumes or multiple nodules. Even DSCs in the 0.60&#x02013;0.65 range can cut annotation time, improve consistency, and lower clinician workload compared to current FDA-approved semi-automated tools, which still require expert-drawn contours (Tessler and Thomas, <xref ref-type="bibr" rid="B17">2023</xref>).</p></sec>
<sec sec-type="conclusions" id="s5">
<title>5 Conclusion</title>
<p>This study presents a transformer-based approach to thyroid nodule segmentation in ultrasound images, leveraging MAE pre-training to accelerate convergence and enhance feature learning. On three public datasets, our model achieved Dice Similarity Coefficients of 0.63 (AIMI), 0.64 (TN3K), and 0.65 (DDTI), demonstrating the feasibility of self-supervised pre-training even with limited annotated data.</p>
<p>Incorporating a cross-attention module did not yield consistent accuracy gains&#x02014;likely because those layers were not pre-trained. Although smaller patch sizes improved reconstruction quality, they also introduced mosaic artifacts, increased context length, and added parameter complexity, resulting in longer convergence times without boosting segmentation performance.</p>
<p>Moving forward, we will integrate boundary-aware loss functions and adopt more extensive data-augmentation strategies to better delineate irregular nodule borders, and we will assemble larger, more diverse datasets to mitigate small-sample limitations. Even at moderate DSC levels, our MAE-driven auto-segmentation pipeline holds promise for reducing the manual delineation workload of radiographers and radiologists, thereby enabling more efficient and scalable clinical workflows.</p></sec>
</body>
<back>
<sec sec-type="data-availability" id="s6">
<title>Data availability statement</title>
<p>Publicly available datasets were analyzed in this study. This data can be found at: <ext-link ext-link-type="uri" xlink:href="https://stanfordaimi.azurewebsites.net/datasets/a72f2b02-7b53-4c5d-963c-d7253220bfd5">https://stanfordaimi.azurewebsites.net/datasets/a72f2b02-7b53-4c5d-963c-d7253220bfd5</ext-link>, <ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/eiraoi/thyroidultrasound">https://www.kaggle.com/datasets/eiraoi/thyroidultrasound</ext-link>, and <ext-link ext-link-type="uri" xlink:href="https://github.com/haifangong/TRFE-Net-for-thyroid-nodule-segmentation">https://github.com/haifangong/TRFE-Net-for-thyroid-nodule-segmentation</ext-link>.</p>
</sec>
<sec sec-type="author-contributions" id="s7">
<title>Author contributions</title>
<p>YX: Visualization, Validation, Writing &#x02013; original draft, Methodology, Writing &#x02013; review &#x00026; editing. RA: Methodology, Writing &#x02013; review &#x00026; editing, Supervision, Investigation, Validation. QL: Resources, Investigation, Writing &#x02013; review &#x00026; editing, Supervision, Validation. JHT: Writing &#x02013; review &#x00026; editing, Supervision, Writing &#x02013; original draft, Software, Resources, Project administration, Validation, Visualization, Methodology, Investigation. C-LC: Funding acquisition, Resources, Writing &#x02013; review &#x00026; editing, Validation, Supervision.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research and/or publication of this article. This work was supported by Duke-NUS Medical School (grant number 03/FY2024/P1/09-A24). The grant recipient is Chiaw-Ling Chng. The work was supported by Clinical &#x00026; Systems Innovation Main Grant (grant number 03/FY2024/P2/03-A125).</p>
</sec>
<ack><p>We would like to thank the AIMI Center at Stanford University, the TN3K dataset creators, and the DDTI dataset team for making their datasets publicly available for research purposes.</p>
</ack>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="ai-statement" id="s9">
<title>Generative AI statement</title>
<p>The author(s) declare that Gen AI was used in the creation of this manuscript. Generative AI was used to polish the language and improve the clarity of expression in the manuscript. The core ideas, analyses, and conclusions are entirely the author(s)&#x00027; own.</p></sec>
<sec sec-type="disclaimer" id="s10">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bi</surname> <given-names>H.</given-names></name> <name><surname>Cai</surname> <given-names>C.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>Y.</given-names></name> <name><surname>Lu</surname> <given-names>G.</given-names></name> <name><surname>Shu</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2023</year>). <article-title>Bpat-unet: boundary preserving assembled transformer unet for ultrasound thyroid nodule segmentation</article-title>. <source>Comput. Methods Programs Biomed</source>. <volume>238</volume>:<fpage>107614</fpage>. <pub-id pub-id-type="doi">10.1016/j.cmpb.2023.107614</pub-id><pub-id pub-id-type="pmid">37244233</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Yu</surname> <given-names>Q.</given-names></name> <name><surname>Luo</surname> <given-names>X.</given-names></name> <name><surname>Adeli</surname> <given-names>E.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Transunet: transformers make strong encoders for medical image segmentation</article-title>. <source>arXiv</source> [Preprint]. arXiv:2102.04306. <pub-id pub-id-type="doi">10.48550/arXiv.2102.04306</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dean</surname> <given-names>D. S.</given-names></name> <name><surname>Gharib</surname> <given-names>H.</given-names></name></person-group> (<year>2008</year>). <article-title>Epidemiology of thyroid nodules</article-title>. <source>Best Pract. Res. Clin. Endocrinol. Metab</source>. <volume>22</volume>, <fpage>901</fpage>&#x02013;<lpage>911</lpage>. <pub-id pub-id-type="doi">10.1016/j.beem.2008.09.019</pub-id><pub-id pub-id-type="pmid">19041821</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A.</given-names></name></person-group> (<year>2020</year>). <article-title>An image is worth 16x16 words: transformers for image recognition at scale</article-title>. <source>arXiv</source> [Preprint]. arXiv:2010.11929. <pub-id pub-id-type="doi">10.48550/arXiv.2010.11929</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Felzenszwalb</surname> <given-names>P. F.</given-names></name> <name><surname>Huttenlocher</surname> <given-names>D. P.</given-names></name></person-group> (<year>2004</year>). <article-title>Efficient graph-based image segmentation</article-title>. <source>Int. J. Comput. Vis</source>. <volume>59</volume>, <fpage>167</fpage>&#x02013;<lpage>181</lpage>. <pub-id pub-id-type="doi">10.1023/B:VISI.0000022288.19776.77</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gong</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>G.</given-names></name> <name><surname>Wang</surname> <given-names>R.</given-names></name> <name><surname>Xie</surname> <given-names>X.</given-names></name> <name><surname>Mao</surname> <given-names>M.</given-names></name> <name><surname>Yu</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Multi-task learning for thyroid nodule segmentation with thyroid region prior,&#x0201D;</article-title> in <italic>2021 IEEE 18th International Symposium on Biomedical Imaging (ISBI)</italic> (NICE: IEEE), <fpage>257</fpage>&#x02013;<lpage>261</lpage>. <pub-id pub-id-type="doi">10.1109/ISBI48211.2021.9434087</pub-id><pub-id pub-id-type="pmid">36812810</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Hart</surname> <given-names>P. E.</given-names></name> <name><surname>Stork</surname> <given-names>D. G.</given-names></name> <name><surname>Duda</surname> <given-names>R. O.</given-names></name></person-group> (<year>2000</year>). <source>Pattern Classification</source>. <publisher-loc>Hoboken, NJ</publisher-loc>: <publisher-name>Wiley</publisher-name>.</citation>
</ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Chen</surname> <given-names>X.</given-names></name> <name><surname>Xie</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Doll&#x000E1;r</surname> <given-names>P.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name></person-group> (<year>2022</year>). &#x0201C;Masked autoencoders are scalable vision learners,&#x0201D; <italic>in Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition</italic> (New Orleans, LA), <fpage>16000</fpage>&#x02013;<lpage>16009</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR52688.2022.01553</pub-id></citation>
</ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>Q.-H.</given-names></name> <name><surname>Lee</surname> <given-names>S.-Y.</given-names></name> <name><surname>Liu</surname> <given-names>L.-Z.</given-names></name> <name><surname>Lu</surname> <given-names>M.-H.</given-names></name> <name><surname>Jin</surname> <given-names>L.-W.</given-names></name> <name><surname>Li</surname> <given-names>A.-H.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>A robust graph-based segmentation method for breast tumors in ultrasound images</article-title>. <source>Ultrasonics</source> <volume>52</volume>, <fpage>266</fpage>&#x02013;<lpage>275</lpage>. <pub-id pub-id-type="doi">10.1016/j.ultras.2011.08.011</pub-id><pub-id pub-id-type="pmid">21925692</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Long</surname> <given-names>J.</given-names></name> <name><surname>Shelhamer</surname> <given-names>E.</given-names></name> <name><surname>Darrell</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). <article-title>&#x0201C;Fully convolutional networks for semantic segmentation,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Boston, MA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3431</fpage>&#x02013;<lpage>3440</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2015.7298965</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oktay</surname> <given-names>O.</given-names></name> <name><surname>Schlemper</surname> <given-names>J.</given-names></name> <name><surname>Folgoc</surname> <given-names>L. L.</given-names></name> <name><surname>Lee</surname> <given-names>M.</given-names></name> <name><surname>Heinrich</surname> <given-names>M.</given-names></name> <name><surname>Misawa</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Attention u-net: learning where to look for the pancreas</article-title>. <source>arXiv</source> [Preprint]. arXiv:1804.03999. <pub-id pub-id-type="doi">10.48550/arXiv.1804.03999</pub-id><pub-id pub-id-type="pmid">35474556</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pedraza</surname> <given-names>L.</given-names></name> <name><surname>Vargas</surname> <given-names>C.</given-names></name> <name><surname>Narv&#x000E1;ez</surname> <given-names>F.</given-names></name> <name><surname>Dur&#x000E1;n</surname> <given-names>O.</given-names></name> <name><surname>Mu&#x000F1;oz</surname> <given-names>E.</given-names></name> <name><surname>Romero</surname> <given-names>E.</given-names></name></person-group> (<year>2015</year>). &#x0201C;An open access thyroid ultrasound image database,&#x0201D; in <italic>10th International Symposium on Medical Information Processing and Analysis, Volume 9287</italic> (Bellingham, WA: SPIE), <fpage>188</fpage>&#x02013;<lpage>193</lpage>. <pub-id pub-id-type="doi">10.1117/12.2073532</pub-id></citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Petit</surname> <given-names>O.</given-names></name> <name><surname>Thome</surname> <given-names>N.</given-names></name> <name><surname>Rambour</surname> <given-names>C.</given-names></name> <name><surname>Themyr</surname> <given-names>L.</given-names></name> <name><surname>Collins</surname> <given-names>T.</given-names></name> <name><surname>Soler</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2021</year>). &#x0201C;U-net transformer: self and cross attention for medical image segmentation,&#x0201D; in <italic>Machine Learning in Medical Imaging: 12th International Workshop, MLMI 2021, Held in Conjunction with MICCAI 2021, Strasbourg, France, September 27, 2021, Proceedings 12</italic> (Cham: Springer), <fpage>267</fpage>&#x02013;<lpage>276</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-030-87589-3_28</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Radhachandran</surname> <given-names>A.</given-names></name> <name><surname>Kinzel</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>J.</given-names></name> <name><surname>Sant</surname> <given-names>V.</given-names></name> <name><surname>Patel</surname> <given-names>M.</given-names></name> <name><surname>Masamed</surname> <given-names>R.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>A multitask approach for automated detection and segmentation of thyroid nodules in ultrasound images</article-title>. <source>Comput. Biol. Med</source>. <volume>170</volume>:<fpage>107974</fpage>. <pub-id pub-id-type="doi">10.1016/j.compbiomed.2024.107974</pub-id><pub-id pub-id-type="pmid">38244471</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Raffel</surname> <given-names>C.</given-names></name> <name><surname>Shazeer</surname> <given-names>N.</given-names></name> <name><surname>Roberts</surname> <given-names>A.</given-names></name> <name><surname>Lee</surname> <given-names>K.</given-names></name> <name><surname>Narang</surname> <given-names>S.</given-names></name> <name><surname>Matena</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Exploring the limits of transfer learning with a unified text-to-text transformer</article-title>. <source>J. Mach. Learn. Res</source>. <volume>21</volume>, <fpage>1</fpage>&#x02013;<lpage>67</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ronneberger</surname> <given-names>O.</given-names></name> <name><surname>Fischer</surname> <given-names>P.</given-names></name> <name><surname>Brox</surname> <given-names>T.</given-names></name></person-group> (<year>2015</year>). &#x0201C;U-net: convolutional networks for biomedical image segmentation,&#x0201D; in <italic>Medical Image Computing and Computer-Assisted Intervention-MICCAI 2015: 18th INTERNATIONAL CONFERENCE, Munich, Germany, October 5-9, 2015, Proceedings, part III 18</italic> (<publisher-loc>Cham</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>234</fpage>&#x02013;<lpage>241</lpage>. <pub-id pub-id-type="doi">10.1007/978-3-319-24574-4_28</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tessler</surname> <given-names>F. N.</given-names></name> <name><surname>Thomas</surname> <given-names>J.</given-names></name></person-group> (<year>2023</year>). <article-title>Artificial intelligence for evaluation of thyroid nodules: a primer</article-title>. <source>Thyroid</source> <volume>33</volume>, <fpage>150</fpage>&#x02013;<lpage>158</lpage>. <pub-id pub-id-type="doi">10.1089/thy.2022.0560</pub-id><pub-id pub-id-type="pmid">36424829</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaswani</surname> <given-names>A.</given-names></name></person-group> (<year>2017</year>). &#x0201C;Attention is all you need,&#x0201D; in <italic>Advances in Neural Information Processing Systems</italic>.</citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Yuan</surname> <given-names>J.</given-names></name> <name><surname>Cheng</surname> <given-names>Q.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Carson</surname> <given-names>P. L.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Medical breast ultrasound image segmentation by machine learning</article-title>. <source>Ultrasonics</source> <volume>91</volume>, <fpage>1</fpage>&#x02013;<lpage>9</lpage>. <pub-id pub-id-type="doi">10.1016/j.ultras.2018.07.006</pub-id><pub-id pub-id-type="pmid">30029074</pub-id></citation></ref>
</ref-list>
</back>
</article>