<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2022.878718</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A Multi-Scale Densely Connected Convolutional Neural Network for Automated Thyroid Nodule Classification</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Wang</surname> <given-names>Luoyan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1606059/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Zhou</surname> <given-names>Xiaogen</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Nie</surname> <given-names>Xingqing</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1605954/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Lin</surname> <given-names>Xingtao</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1606024/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Li</surname> <given-names>Jing</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Zheng</surname> <given-names>Haonan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Xue</surname> <given-names>Ensheng</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Shun</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Chen</surname> <given-names>Cong</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Du</surname> <given-names>Min</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Tong</surname> <given-names>Tong</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/556226/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Gao</surname> <given-names>Qinquan</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1698082/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zheng</surname> <given-names>Meijuan</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="corresp" rid="c002"><sup>&#x0002A;</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Physics and Information Engineering, Fuzhou University</institution>, <addr-line>Fuzhou</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>Fujian Key Lab of Medical Instrumentation &#x00026; Pharmaceutical Technology, Fuzhou University</institution>, <addr-line>Fuzhou</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>Fujian Medical University Union Hospital</institution>, <addr-line>Fuzhou</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Fujian Medical Ultrasound Research Institute</institution>, <addr-line>Fuzhou</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Mufti Mahmud, Nottingham Trent University, United Kingdom</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Lidong Yang, Inner Mongolia University of Science and Technology, China; Runzhi Li, Zhengzhou University, China</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Qinquan Gao <email>gqinquan&#x00040;fzu.edu.cn</email></corresp>
<corresp id="c002">Meijuan Zheng <email>502218654&#x00040;qq.com</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Brain Imaging Methods, a section of the journal Frontiers in Neuroscience</p></fn></author-notes>
<pub-date pub-type="epub">
<day>19</day>
<month>05</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2022</year>
</pub-date>
<volume>16</volume>
<elocation-id>878718</elocation-id>
<history>
<date date-type="received">
<day>18</day>
<month>02</month>
<year>2022</year>
</date>
<date date-type="accepted">
<day>13</day>
<month>04</month>
<year>2022</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2022 Wang, Zhou, Nie, Lin, Li, Zheng, Xue, Chen, Chen, Du, Tong, Gao and Zheng.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>Wang, Zhou, Nie, Lin, Li, Zheng, Xue, Chen, Chen, Du, Tong, Gao and Zheng</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Automated thyroid nodule classification in ultrasound images is an important way to detect thyroid nodules and to make a more accurate diagnosis. In this paper, we propose a novel deep convolutional neural network (CNN) model, called n-ClsNet, for thyroid nodule classification. Our model consists of a multi-scale classification layer, multiple skip blocks, and a hybrid atrous convolution (HAC) block. The multi-scale classification layer first obtains multi-scale feature maps in order to make full use of image features. After that, each skip-block propagates information at different scales to learn multi-scale features for image classification. Finally, the HAC block is used to replace the downpooling layer so that the spatial information can be fully learned. We have evaluated our n-ClsNet model on the TNUI-2021 dataset. The proposed n-ClsNet achieves an average accuracy (ACC) score of 93.8% in the thyroid nodule classification task, which outperforms several representative state-of-the-art classification methods.</p></abstract>
<kwd-group>
<kwd>the thyroid nodule classification</kwd>
<kwd>multi-scale</kwd>
<kwd>densely connection</kwd>
<kwd>hybrid atrous convolution</kwd>
<kwd>deep convolutional neural network</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="4"/>
<equation-count count="10"/>
<ref-count count="37"/>
<page-count count="12"/>
<word-count count="6799"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>Proper balancing of hormones, which regulates metabolism in the human body, is a main sign to identify the healthy nature of human beings. The tyroid gland is responsible for balancing hormones in human being. Therefore, the thyroid is an essential butterfly shaped organ which is positioned in front of the neck (Gulame et al., <xref ref-type="bibr" rid="B10">2021</xref>). A thyroid nodule is a discrete lesion within the thyroid gland that is radiologically distinct from the surrounding thyroid parenchyma (Haugen et al., <xref ref-type="bibr" rid="B11">2016</xref>). Thyroid nodules are very common in the general population. About 19&#x02013;68% of individuals are detected to have thyroid nodules with high resolution ultrasound imaging (Liu et al., <xref ref-type="bibr" rid="B19">2019</xref>). Generally, only nodules &#x0003E;1 cm should be evaluated, since they have a greater potential to be clinically significant cancers. In very rare cases, some nodules &#x0003C;1 cm yet may cause future morbidity and mortality Haugen et al. (<xref ref-type="bibr" rid="B11">2016</xref>). Thyroid cancer accounts for 3% of the global incidence of all cancers, with 586,000 new patients estimated in Miranda-Filho et al. (<xref ref-type="bibr" rid="B22">2021</xref>). Since ultrasound image provides a non-invasive and realtime inspection at a low cost, ultrasonography has become the best selection for the clinical identification of thyroid nodules (Gulame et al., <xref ref-type="bibr" rid="B10">2021</xref>). However, due to ultrasound image being influenced by echo and speckle noise, experienced radiologists usually diagnose based on the shape, margin, and boundaries of sonographic characteristics of nodules in ultrasound image slices. It is fairly subjective and extremely dependent upon the clinical experience of radiologists (Yang et al., <xref ref-type="bibr" rid="B35">2021</xref>). So as to handle this challenge, computer aided classification using ultrasound images is quite important in thyroid nodule identification. The automatic classification of thyroid nodules can differentiate whether a nodule is benign or malignant, which reduces the workload and inexperienced young radiologists&#x00027; misdiagnosis rate (Yang et al., <xref ref-type="bibr" rid="B35">2021</xref>).</p>
<p>Two steps are required in machine learning (ML)-based methods for thyroid nodule classification. Features are first extracted and then a classifier is built to perform an automated classification. For instance, random forest Ouyang et al. (<xref ref-type="bibr" rid="B23">2019</xref>), backpropagation neural network (BPNN) Kumari and Rani (<xref ref-type="bibr" rid="B18">2019</xref>), and stationary wavelet transform Acharya et al. (<xref ref-type="bibr" rid="B3">2014</xref>) have been well applied in the classification of thyroid nodules.</p>
<p>In recent years, deep learning models have successfully been used in image classification tasks, since they have shown superior performance to conventional learning methods. One benefit of deep learning is that it can extract deep features hidden in the sonographic image that human radiologists may not visually inspect. In addition, it can integrate the feature extraction and classification into a uniform framework, which avoids the processes of complex hand-crafted features extraction and classifier selection (Yang et al., <xref ref-type="bibr" rid="B35">2021</xref>). Therefore, various deep learning-based methods have been proposed for different classification tasks.</p>
<p>For the classification of natural images, Krizhevsky et al. (<xref ref-type="bibr" rid="B17">2012</xref>) presented a groundbreaking networks, which demonstrated that deep learning models have superior performance in the classification domain. Szegedy et al. (<xref ref-type="bibr" rid="B31">2015</xref>) used a deep convolutional neural network (CNN) architecture called Inception, which obtained further improvements in image classification over AlexNet (Krizhevsky et al., <xref ref-type="bibr" rid="B17">2012</xref>). Simonyan and Zisserman (<xref ref-type="bibr" rid="B28">2014</xref>) proposed a very deep convolutional network, which moved a step forward to deepen networks in image recognition. He et al. (<xref ref-type="bibr" rid="B12">2016</xref>) proposed an extraordinary structure referred to as ResNet, which solved the degradation problem in an extremely deep convolutional network. Iandola et al. (<xref ref-type="bibr" rid="B16">2016</xref>) proposed a lightweight CNN architecture called SqueezeNet to speed up the inference process without loosing accuracy (ACC). Howard et al. (<xref ref-type="bibr" rid="B13">2017</xref>) presented a lightweight and efficient neural network, which can achieve a high ACC of classification. Huang et al. (<xref ref-type="bibr" rid="B14">2017</xref>) introduced the dense convolutional network to strengthen feature propagation and encourage feature reuse.</p>
<p>Several studies based on deep neural networks have been carried out for the classification of thyroid nodules (Song et al., <xref ref-type="bibr" rid="B29">2018</xref>; Zhang et al., <xref ref-type="bibr" rid="B36">2019a</xref>). However, the use of classification models in a natural image may lead to poor generalization problems. First of all, since the amount of natural images is much larger than the number of medical images, it is difficult to achieve the same ACC for the classification of thyroid nodules on the Caltech-101 dataset. Different from natural images, it is difficult to obtain millions of ultrasound images in clinical practice. Therefore, it is a challenge to train deep learning models using a small set of ultrasound images for the classification of thyroid nodules. In clinical practice, experienced radiologists distinguish whether a thyroid nodule is benign or malignant in ultrasound image slices <italic>via</italic> visual inspections. However, the process is not only time consuming and has high labor cost, but also has extremely subjective biases.</p>
<p>Based on the advantages of CNN and Transformer, we propose an O-Net framework to combine the CNN and the Transformer to learn both global and local contextual features. We combine the CNN and Swin Transformer as encoder first and send them into a CNN-based decoder and a Swin Transformer-based decoder, respectively. The results of two decoders are fused to get the final result. This network combines the advantages of CNN and Transformer and may improve the performance of medical image segmentation. Our experimental results have shown that the performance of the network can be significantly improved by combining CNN and Transformer. In addition, a classification task is simultaneously performed based on the O-Net. Experiments show that the segmentation results are beneficial for improving the ACC of the classification task. Experiments on the Synapse multi-organ segmentation dataset and the ISIC2017 skin lesion challenge dataset have demonstrated the superiority of our method compared to other state-of-the-art segmentation methods. In addition, based on the segmentation network, the performance of the classification network has also been greatly improved.</p>
<p>Data is the key to the performance of classification networks based on deep learning. Classification networks that show good performance in natural images are difficult to achieve the same high ACC in medical image classification. Most of the existing thyroid nodule classification methods use the natural image classification network as the backbone network architecture. However, the classification network used for natural images does not fully adapt to medical images because the number of thyroid nodule images is far less than that of natural images. In the case of small amount of data, there is a risk of overfitting when deep network suitable for natural images is used to classify thyroid nodules. Therefore, this paper proposes a new method to solve this problem.</p>
<p>The method includes the following steps. First, we adopt a multi-scale input layer to excavate multiscale features. Then, we design specialized skip-block exploit depth features. Finally, we employ hybrid atrous convolution block substitute downsampling. In general, the main innovations of this paper include the follwing:</p>
<p>1) We design a skip-block as depth feature extractor, which consists of convolution layer, batch normalization layer, skip connection layer, and activation function. This skip-block is used to learn the deep features of thyroid nodules. Its skip connection structure deepens the network while reducing the risk of overfitting.</p>
<p>2) We propose a novel hybrid atrous convolution (HAC) block substitute downsampling in order to reduce the loss of spatial information caused by downsampling. This framework with HAC effectively enlarges the receptive fields of the network to aggregate global information.</p>
</sec>
<sec id="s2">
<title>2. Related Works</title>
<sec>
<title>2.1. Thyroid Nodule Classification Based on ML</title>
<p>Computer-aided diagnostic (CAD) system of thyroid nodules has a long history. For objective differentiation of benign/malignant thyroid lesions, various CAD systems based on ML have been exploited (Chang et al., <xref ref-type="bibr" rid="B5">2010</xref>, <xref ref-type="bibr" rid="B6">2016</xref>; Iakovidis et al., <xref ref-type="bibr" rid="B15">2010</xref>; Acharya et al., <xref ref-type="bibr" rid="B1">2011</xref>, <xref ref-type="bibr" rid="B2">2012</xref>; Ding et al., <xref ref-type="bibr" rid="B7">2011</xref>; Raghavendra et al., <xref ref-type="bibr" rid="B26">2017</xref>; Ardakani et al., <xref ref-type="bibr" rid="B4">2018</xref>; Prochazka et al., <xref ref-type="bibr" rid="B24">2019a</xref>,<xref ref-type="bibr" rid="B25">b</xref>; Lu et al., <xref ref-type="bibr" rid="B21">2020</xref>).</p>
<p>Earlier ML approaches for thyroid nodule classification include two steps: hand-crafted features are first extracted and then used in the support vector machine (SVM) or k-nearest neighbor (KNN) classifier to build the automated classification system for the diagnosis of malignant thyroid nodules (Chang et al., <xref ref-type="bibr" rid="B5">2010</xref>; Iakovidis et al., <xref ref-type="bibr" rid="B15">2010</xref>; Acharya et al., <xref ref-type="bibr" rid="B1">2011</xref>; Ding et al., <xref ref-type="bibr" rid="B7">2011</xref>).</p>
<p>Afterward, the CAD system used for thyroid nodules tries to consider combining various features and different classifiers. In another study by Acharya et al. (<xref ref-type="bibr" rid="B2">2012</xref>), integrated features include the following: local binary pattern, laws texture energy, Fourier descriptor, and Fourier spectrum descriptor, using ultrasound images of 20 nodules (10 benign images and 10 malignant images) to extract features. Then, resulting feature vectors were used to build seven different classifiers in order to compare the performances, including SVM, decision tree, sugeno fuzzy, gaussian mixture model (GMM), KNN, radial basis probabilistic neural network, and naive Bayes classifier. The result shows that SVM and fuzzy classifier achieved the highest classification ACC of 100%, whereas, the GMM classifier peaked at an ACC of 98%. Chang et al. (<xref ref-type="bibr" rid="B6">2016</xref>) employed histogram, intensity differences, elliptical fit, gray-level co-occurrence matrices, and gray-level run-length matrices to abstract features from 30 malignant and 29 benign images, which then used SVM classifier and leaveone-out cross-validation to differentiate benign and malignant nodules, consequently achieving ACC of 98.4%. Raghavendra et al. (<xref ref-type="bibr" rid="B26">2017</xref>) proposed the CAD based on a binary stack decomposition algorithm, which extracted 181 features from 242 images and achieved ACC of 97.5% using SVM classifier. Therefore, we can conclude that selecting features and then constructing a classifier is very important for thyroid nodule classification, which is the key to promoting classification ACC.</p>
<p>From more recently published studies, how to extract more features from the original image and select features carefully is still the key to thyroid nodule classification based on ML. Ardakani et al. (<xref ref-type="bibr" rid="B4">2018</xref>) proposed the CAD based on textural and morphological features, which is capable of distinguishing thyroid nodules from ultrasound images by utilizing a support vector machine classifier. Prochazka et al. (<xref ref-type="bibr" rid="B25">2019b</xref>) designed a CAD that divided 60 thyroid nodules (20 malignant images, 40 benign images) into small patches of 17 &#x000D7; 17 pixels, which were used to extract several direction independent features by employing two-threshold binary decomposition. The features were used in random forests (RF) and SVM classifiers to categorize nodules into malignant and benign classes, then obtained the ACC score of 91.6%. In another study, Prochazka et al. (<xref ref-type="bibr" rid="B24">2019a</xref>) applied histogram analysis and segmentation-based fractal texture analysis algorithm, which calculated direction-independent features only. The features were used in SVM and RF classifiers to differentiate nodules into malignant and benign classes. Using the leave-one-out cross-validation method, the overall ACC was 92.42% for RF and 94.64% for SVM. Lu et al. (<xref ref-type="bibr" rid="B21">2020</xref>) extracted shape features, texture features, and local binary pattern features from original ultrasound images (59 patients). Then, the multi-kernel support vector machine classifier was configured with 10 linear kernels to combine features from different categories for classification, achieving the best ACC for the sub-class at 94.44%.</p>
</sec>
<sec>
<title>2.2. Thyroid Nodule Classification Based on Deep Learning</title>
<p>The classical ML algorithms usually require complex feature engineering, which first selects features and then uses it in classifier. However, the deep learning only needs to pass the data directly to the neural networks. Thus, one of the most growing trends of ML is deep learning (Sharifi et al., <xref ref-type="bibr" rid="B27">2021</xref>). Song et al. (<xref ref-type="bibr" rid="B29">2018</xref>) developed a cascade convolution neural network framework, which confirmed the feasibility of CNN used for thyroid nodules detection and recognition. Zhang et al. (<xref ref-type="bibr" rid="B36">2019a</xref>) adopted a tripartite classification module based on CNN model to pick out nodules information in ultrasound images. Wang et al. (<xref ref-type="bibr" rid="B32">2020</xref>) proposed a dual-attention ResNet-based classification network to automatically achieve the accurate classification of thyroid nodules. Specifically, they adopted ResNet200 as the backbone network architecture to perform the classification of thyroid nodules while there is the problem that the classification network used for natural images does not fully adapt to medical images.</p>
</sec>
</sec>
<sec id="s3">
<title>3. The Proposed Method</title>
<sec>
<title>3.1. Overall Architecture Design</title>
<p>In this paper, we proposed a n-ClsNet classification model, which consists of a multi-scale classification layer, skip blocks, and HAC block. The multi-scale classification layer supported the n-ClsNet model capture several scale features on small-scale dataset. In an insufficient data case, tackled various features are quite important to classification networks. In each skip block, the convolution with skip connection can handle several scale information from a multiscale layer. Our proposed n-ClsNet is specialized for benign and malignant binary classification tasks of thyroid nodules. The n-ClsNet network&#x00027;s framework is shown in <xref ref-type="fig" rid="F1">Figure 1</xref>.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Illustration of the architecture of the proposed n-ClsNet network, which consists of two blocks: Skip-block and hybrid atrous convolution (HAC) block.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-878718-g0001.tif"/>
</fig>
</sec>
<sec>
<title>3.2. Image Preprocess</title>
<p>In this paragraph, we foremost introduce the data augmentation and image pretreatment strategies, which are used in training and testing stages. Due to the limited number of medical image datasets, the datasets are enlarged to reduce the hazard of overfitting (Sun et al., <xref ref-type="bibr" rid="B30">2018</xref>). For image preprocessing, we first select several transformations, includes vertical flip, horizontal flip and rotation. The direction of rotation including 45/90/135/180/225/270/315. Besides, we employ noise interference, which selected gauss noise. The visualization of transformation is shown in <xref ref-type="fig" rid="F2">Figure 2</xref>.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Examples of several transformations for thyroid nodule image.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-878718-g0002.tif"/>
</fig>
</sec>
<sec>
<title>3.3. Multi-Scale Classification Framework</title>
<p>The multi-scale input layer was extensively used for the segmentation of images. The deep learning model that adopted multi-scale input has been demonstrated to increase the performance of segmentation (Gu et al., <xref ref-type="bibr" rid="B9">2019</xref>). Because multi-scale input could integrate various information from feature maps to avoid the large parameters in follow-up networks. Not only that, multi-scale input could enlarge the network width. Therefore, the above advantages of multi-scale input can be applied not only to image segmentation algorithms but also to image classification algorithms. Then, we determine to introduce a multi-scale method into the n-ClsNet to achieve supplementary feature representation in scale interspace. Different from Fu et al. (<xref ref-type="bibr" rid="B8">2018</xref>), they pushed the multiscale feature map to multi-scream networks and concatenate the ultimate feature map in the last layer, we employ the max-pooling layer to downsampling the image effectively and construct the feature detectors with different receptive field sizes. In our multi-scale classification framework, we first adopt downsampling of different multiples in order to obtain image patches sample of different sizes. According to the size of the original image from thyroid nodules, we design four branches downsampling with four scales. In each branch, the thyroid nodules image is followed by feature extractors in order that receiving abundant information on image features. Then, each branch connects to the former branch. Specifically, the first branch only has a depth feature extractor while others have both shallow feature extractor and depth feature extractor. The shallow feature extractor from the last three branches was combined with the depth feature extractor from the former branch. This method is more advantageous for characterizing diverse size structures in ultrasound imaging than single scale framework. The architecture of this multi-scale classification framework is shown in <xref ref-type="fig" rid="F3">Figure 3</xref>.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Schematic diagram of our proposed multi-scale classification framework for thyroid nodules.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-878718-g0003.tif"/>
</fig>
</sec>
<sec>
<title>3.4. Skip Block</title>
<p>In our n-ClsNet framework, we design the skip-block as a depth feature extractor, which receives shallow feature map from the former convolution layer. This skip-block is composed of a convolution layer, batch normalization layer, skip connection layer, and activation function. With regard to convolution layer from skip-block, we select 3 &#x000D7; 3 convolutional kernels and stuff a layer of edge pixels. They are followed by a batch normalization layer in order to alleviate the disappearance of gradients. After that, the ReLU is used as the activation function, which introduced a non-linear element to further overcome the problem of vanishing gradient. The skip connection layer is designed to leapfrog the structure composed of the convolutional layer, batch normalization layer, and ReLU. This layer is directly connected to a straight link that pushed the output of the shallow feature extractor and pushed into the ReLU. In order to conform to the demand of thyroid nodules datasets, we tried two kinds of skip connection layers and compared the residual architecture of the ResNet34-layer (Simonyan and Zisserman, <xref ref-type="bibr" rid="B28">2014</xref>). This residual architecture is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. One version of the skip connection layer only has one convolutional layer, which is designed as 1 &#x000D7; 1 convolutional kernels to change the number of channels. The structure of this version is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. The other version of the skip connection layer has both convolutional layer and batch normalization layer, the configuration of convolutional layer is the same as the former. The structure of our choosen version is shown in <xref ref-type="fig" rid="F4">Figure 4</xref>. In our research, the skip connection layer increased the batch normalization layer, which promoted the quality of classification effectively for thyroid nodules. Specifically, we conducted a comparative experiment to prove this viewpoint.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The illustrations of skip connection layer and residual architecture of resnet34-layer. Among them, <bold>(A, B)</bold> are the versions of skip connection layer designed by us, and <bold>(C)</bold> is the residual architecture of ResNet34-layer.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-878718-g0004.tif"/>
</fig>
</sec>
<sec>
<title>3.5. HAC Block</title>
<p>In order to reduce the loss of spatial information caused by downsampling, we employed dilated convolution substitute downsampling in the model (Liu et al., <xref ref-type="bibr" rid="B20">2017</xref>). Besides, dilated convolution can increase receptive field size, even will not reduce the spatial resolution of the intermediate feature map. The dilated convolution can be described as follows:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>Y</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munderover accentunder="false" accent="false"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover></mml:mstyle><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>r</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>Y</italic>() is the output feature map, <italic>F</italic><sub><italic>i</italic></sub>() is the input feature map, <italic>p</italic> is the processing pixel, <italic>n</italic> is the pixel used in the convolution process, and <italic>r</italic> is the dilation rate. The dilation rate depends on the stride of the input feature map. The dilated convolution is commonly available in two connection of types called parallel type and cascade type (Xiong et al., <xref ref-type="bibr" rid="B34">2021</xref>). The HAC has parallel mode and cascade mode. In a word, the output feature map consists of four aisles atrous convolution and the operation of dimension mapping. Specifically, the output signal of the HAC block is defined as:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>H</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02225;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02225;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x02225;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mn>4</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mo>&#x02225;</mml:mo><mml:mi>D</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mn>4</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mn>3</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mn>7</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>,</mml:mo></mml:mtd></mml:mtr><mml:mtr></mml:mtr></mml:mtable></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>In our HAC block, we adopt three atrous convolutions. The architecture of these three atrous convolutions is shown in <xref ref-type="fig" rid="F5">Figure 5</xref>. Due to the fact that we choose atrous convolution with four aisles, where <italic>A</italic><sub><italic>c</italic>1</sub>(<italic>F</italic><sub><italic>i</italic></sub>) is the first aisle for atrous convolution, <italic>A</italic><sub><italic>r</italic>1</sub>(<italic>F</italic><sub><italic>i</italic></sub>), <italic>A</italic><sub><italic>r</italic>3</sub>(<italic>F</italic><sub><italic>i</italic></sub>), and <italic>A</italic><sub><italic>r</italic>7</sub>(<italic>F</italic><sub><italic>i</italic></sub>) is atrous convolution with a learning rate of 1, 3, 7, respectively, <italic>H</italic> is the output feature map, and <italic>D</italic>(<italic>F</italic><sub><italic>i</italic></sub>) is the operation of dimension mapping. Then, we display each aisle atrous convolution in particular, as shown in <xref ref-type="fig" rid="F6">Figure 6</xref>. This framework with HAC effectively enlarges the receptive fields of the network to aggregate global information. In our research, the HAC block with parallel mode and cascade mode have evidently helpful improvement in classification ACC. In the experimental part, we compared the HAC block with the atrous spatial pyramid pooling (ASPP) block to verify the superiority of the HAC block in improving the classification accuracy of thyroid nodules.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>The illustrations of three kinds of atrous convolutions. Left to right: the atrous convolution have dilation rates of <italic>r</italic> = 1, 3, 7, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-878718-g0005.tif"/>
</fig>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>The architecture of HAC with four aisles atrous convolution and the operation of dimension mapping (see the part of white).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-878718-g0006.tif"/>
</fig>
</sec>
</sec>
<sec id="s4">
<title>4. Experiments</title>
<sec>
<title>4.1. Datasets</title>
<p>We employ 2,615 ultrasound thyroid nodule images that were manually labeled by doctors in Fujian Union Medical College Hospital, called the TNUI-2021 dataset, which evaluates the robustness and effectiveness of our classification method. There are 1,834 training samples, 523 validation samples, and 258 testing samples. In addition, the original image size of the TNUI-2021 dataset is 780 &#x000D7; 780, we resize the images to 224 &#x000D7; 224 in the image preprocessing stage.</p>
</sec>
<sec>
<title>4.2. Implementation Details</title>
<p>All models in this experiment were trained on an Ubuntu system with Nvidia RTX 2080TI GPUs. The experiments were performed with SGD as the optimizer, CrossEntropyLoss as the loss, initial learning rate set to 0.1, and weight decay set to 0.001. Two hundred epochs were performed for all experiments.</p>
</sec>
<sec>
<title>4.3. Evaluation Metrics</title>
<p>In order to evaluate the classification performance, several model evaluation indices are used in the experiment, including ACC, Average Precision (AP), area under the receiver operator curve (AUC), Precision, F1-score, and Specificity, which are calculated as follows:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>C</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E5"><label>(5)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E6"><label>(6)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>S</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E7"><label>(7)</label><mml:math id="M7"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E8"><label>(8)</label><mml:math id="M8"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>F</mml:mi><mml:mn>1</mml:mn><mml:mo>-</mml:mo><mml:mi>s</mml:mi><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:mfrac><mml:mo>&#x0002B;</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E9"><label>(9)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>U</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x02208;</mml:mo><mml:mi>p</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>v</mml:mi><mml:mi>e</mml:mi><mml:mtext>&#x00A0;</mml:mtext><mml:mi>c</mml:mi><mml:mi>l</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mfrac><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac></mml:mrow><mml:mrow><mml:mi>M</mml:mi><mml:mo>&#x000D7;</mml:mo><mml:mi>N</mml:mi></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<disp-formula id="E10"><label>(10)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mi>A</mml:mi><mml:mi>P</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true"><mml:munder class="msub"><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:munder></mml:mstyle><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:msub><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mi>R</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:msub><mml:mrow><mml:mi>l</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x000D7;</mml:mo><mml:mi>P</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>TP</italic>, <italic>TN</italic>, <italic>FP</italic>, and <italic>FN</italic> denote the number of true positives, true negatives, false positives, and false negatives, respectively. <italic>M</italic> is the number of positive samples. <italic>N</italic> is the number of negative samples. <italic>rank</italic><sub><italic>i</italic></sub> is the serial number of the i-th sample. The ACC displays the performance of our n-ClsNet model in classifying nodules as malignant or benign. Specificity shows the proportion of correctly identified benign nodules (Wang et al., <xref ref-type="bibr" rid="B33">2018</xref>).</p>
</sec>
<sec>
<title>4.4. Method Comparison</title>
<p>We compare our proposed model with several representative state-of-the-art classification approaches on the TNUI-2021 dataset from the comparison shown in <xref ref-type="table" rid="T1">Table 1</xref>. We compare the proposed n-ClsNet with the state-of-the-art classification algorithms ARL50 (Zhang et al., <xref ref-type="bibr" rid="B37">2019b</xref>) used for medical imaging. In addition, some classical deep learning based classification methods, Alexnet (Krizhevsky et al., <xref ref-type="bibr" rid="B17">2012</xref>), GoogleNet (Szegedy et al., <xref ref-type="bibr" rid="B31">2015</xref>), VGG (Simonyan and Zisserman, <xref ref-type="bibr" rid="B28">2014</xref>), ResNet34 (He et al., <xref ref-type="bibr" rid="B12">2016</xref>), SqueezeNet (Iandola et al., <xref ref-type="bibr" rid="B16">2016</xref>), MobilenetV1 (Howard et al., <xref ref-type="bibr" rid="B13">2017</xref>), and DenseNet (Huang et al., <xref ref-type="bibr" rid="B14">2017</xref>), are also included in the comparison. In order to explain intuitively, we compare of receiver operator curve (ROC)-Accuracy (ACC) curves of nine classification approaches on TNUI-2021 datasets, as shown in <xref ref-type="fig" rid="F7">Figure 7</xref>.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Performance of our method and other methods in classification of thyroid nodules.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>ACC</bold></th>
<th valign="top" align="center"><bold>AP</bold></th>
<th valign="top" align="center"><bold>AUC</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Specificity</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ARL50 (Zhang et al., <xref ref-type="bibr" rid="B37">2019b</xref>)</td>
<td valign="top" align="center">0.8992</td>
<td valign="top" align="center">0.9090</td>
<td valign="top" align="center">0.9343</td>
<td valign="top" align="center">0.8377</td>
<td valign="top" align="center">0.8047</td>
</tr>
<tr>
<td valign="top" align="left">ResNet34 (He et al., <xref ref-type="bibr" rid="B12">2016</xref>)</td>
<td valign="top" align="center">0.8837</td>
<td valign="top" align="center">0.8638</td>
<td valign="top" align="center">0.9218</td>
<td valign="top" align="center">0.8205</td>
<td valign="top" align="center">0.7813</td>
</tr>
<tr>
<td valign="top" align="left">MobilenetV1 (Howard et al., <xref ref-type="bibr" rid="B13">2017</xref>)</td>
<td valign="top" align="center">0.8760</td>
<td valign="top" align="center">0.8474</td>
<td valign="top" align="center">0.8852</td>
<td valign="top" align="center">0.8025</td>
<td valign="top" align="center">0.7500</td>
</tr>
<tr>
<td valign="top" align="left">DenseNet (Huang et al., <xref ref-type="bibr" rid="B14">2017</xref>)</td>
<td valign="top" align="center">0.8837</td>
<td valign="top" align="center">0.9555</td>
<td valign="top" align="center">0.9607</td>
<td valign="top" align="center">0.8247</td>
<td valign="top" align="center">0.7891</td>
</tr>
<tr>
<td valign="top" align="left">SqueezeNet (Iandola et al., <xref ref-type="bibr" rid="B16">2016</xref>)</td>
<td valign="top" align="center">0.8682</td>
<td valign="top" align="center">0.9286</td>
<td valign="top" align="center">0.9375</td>
<td valign="top" align="center">0.8038</td>
<td valign="top" align="center">0.7578</td>
</tr>
<tr>
<td valign="top" align="left">VGG (Simonyan and Zisserman, <xref ref-type="bibr" rid="B28">2014</xref>)</td>
<td valign="top" align="center">0.8488</td>
<td valign="top" align="center">0.9179</td>
<td valign="top" align="center">0.9266</td>
<td valign="top" align="center">0.7758</td>
<td valign="top" align="center">0.7109</td>
</tr>
<tr>
<td valign="top" align="left">GoogleNet (Szegedy et al., <xref ref-type="bibr" rid="B31">2015</xref>)</td>
<td valign="top" align="center">0.8837</td>
<td valign="top" align="center">0.9290</td>
<td valign="top" align="center">0.9439</td>
<td valign="top" align="center">0.8205</td>
<td valign="top" align="center">0.7813</td>
</tr>
<tr>
<td valign="top" align="left">Alexnet (Krizhevsky et al., <xref ref-type="bibr" rid="B17">2012</xref>)</td>
<td valign="top" align="center">0.7907</td>
<td valign="top" align="center">0.7553</td>
<td valign="top" align="center">0.8096</td>
<td valign="top" align="center">0.7184</td>
<td valign="top" align="center">0.6172</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>0.9380</bold></td>
<td valign="top" align="center"><bold>0.9738</bold></td>
<td valign="top" align="center"><bold>0.9756</bold></td>
<td valign="top" align="center"><bold>0.9014</bold></td>
<td valign="top" align="center"><bold>0.8906</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Bold values indicate optimal values</italic>.</p>
</table-wrap-foot>
</table-wrap>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Comparison of receiver operator curve (ROC)-Accuracy (ACC) curves of nine classification approaches on TNUI-2021 datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnins-16-878718-g0007.tif"/>
</fig>
<p>On comparison with the ARL50 (Zhang et al., <xref ref-type="bibr" rid="B37">2019b</xref>), which is attention residual learning convolutional neural network, the <italic>ACC</italic> increases from 0.8992 to 0.938, the <italic>AP</italic> increases from 0.909 to 0.9738, the <italic>AUC</italic> increases from 0.9343 to 0.9756, the <italic>Precision</italic> increases from 0.8377 to 0.9014, and the <italic>Specificity</italic> increases from 0.8047 to 0.8906. For classic algorithms, compared with ResNet34 (He et al., <xref ref-type="bibr" rid="B12">2016</xref>), the <italic>ACC</italic> is increased by 5.4% from 0.8837 to 0.938, the <italic>AP</italic> is increased by 11% from 0.8638 to 0.9738, the <italic>AUC</italic> is increased by 5.3% from 0.9218 to 0.9756, the <italic>Precision</italic> is increased by 8% from 0.8205 to 0.9014, and the <italic>Specificity</italic> is increased by 10% from 0.7813 to 0.8906, respectively. We also compare n-ClsNet with the MobilenetV1 Howard et al. (<xref ref-type="bibr" rid="B13">2017</xref>), the <italic>ACC</italic> increases from 0.876 to 0.938 by 6.2%, the <italic>AP</italic> increases from 0.8474 to 0.9738 by 12%, the <italic>AUC</italic> increases from 0.8852 to 0.9756 by 9%, the <italic>Precision</italic> increases from 0.8025 to 0.9014 by 9%, and the <italic>Specificity</italic> increases from 0.75 to 0.8906 by 14%. For the model evaluation index <italic>ACC</italic>, compared with DenseNet (Huang et al., <xref ref-type="bibr" rid="B14">2017</xref>), the <italic>ACC</italic> increases from 0.8837 to 0.938; compared with SqueezeNet (Iandola et al., <xref ref-type="bibr" rid="B16">2016</xref>), the <italic>ACC</italic> increases from 0.8682 to 0.938; compared with VGG (Simonyan and Zisserman, <xref ref-type="bibr" rid="B28">2014</xref>), the <italic>ACC</italic> increases from 0.8488 to 0.938; compared with GoogleNet (Szegedy et al., <xref ref-type="bibr" rid="B31">2015</xref>), the <italic>ACC</italic> increases from 0.8837 to 0.938; compared with Alexnet (Krizhevsky et al., <xref ref-type="bibr" rid="B17">2012</xref>), the <italic>ACC</italic> increases from 0.7907 to 0.938.</p>
<p>On the thyroid nodule dataset, our model achieves a remarkably higher <italic>ACC</italic>, <italic>AP</italic>, <italic>AUC</italic>, <italic>Precision</italic>, and <italic>Specificity</italic> than others, the highest <italic>ACC</italic> of 0.938, the highest <italic>AP</italic> of 0.9738, the highest <italic>AUC</italic> of 0.9756, the highest <italic>Precision</italic> of 0.9014, and the highest <italic>Specificity</italic> of 0.8906, which proves that our proposed method has a robustness classification ability.</p>
</sec>
<sec>
<title>4.5. Comparison of Skip-Block With Residual Architecture</title>
<sec>
<title>4.5.1. Comparative Experiment for the ResNet34-Residual</title>
<p>To verify the effectiveness of skip-block, we conducted an experiment that skip-block has better performance, in contrast with the residual architecture from ResNet34 (He et al., <xref ref-type="bibr" rid="B12">2016</xref>). A quantitative comparison is shown in <xref ref-type="table" rid="T2">Table 2</xref>. The &#x0201C;No-Batch-Normalization&#x0201D; is one of the versions of skip-block we designed, which has only one convolution layer. The table shows that &#x0201C;No-Batch-Normalization&#x0201D; surpasses &#x0201C;ResNet34-Residual&#x0201D; in the model evaluation of <italic>ACC</italic>, <italic>AP</italic>, <italic>F</italic>1&#x02212;<italic>score</italic>, <italic>Precision</italic>, and <italic>Recall</italic>. From the comparison, our final version skip-block achieves 0.9341, 0.9726, 0.9337, 0.8844, and 0.9336 in <italic>ACC</italic>, <italic>AP</italic>, <italic>F</italic>1&#x02212;<italic>score</italic>, <italic>Precision</italic>, and <italic>Recall</italic>, respectively, better than &#x0201C;ResNet34-Residual.&#x0201D;</p>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Comparison to skip-block and residual architecture.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>ACC</bold></th>
<th valign="top" align="center"><bold>AP</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ResNet34-Residual</td>
<td valign="top" align="center">0.9109</td>
<td valign="top" align="center">0.9224</td>
<td valign="top" align="center">0.9104</td>
<td valign="top" align="center">0.8639</td>
<td valign="top" align="center">0.9103</td>
</tr>
<tr>
<td valign="top" align="left">No-BatchNormalization</td>
<td valign="top" align="center">0.9341</td>
<td valign="top" align="center">0.9726</td>
<td valign="top" align="center">0.9337</td>
<td valign="top" align="center">0.8844</td>
<td valign="top" align="center">0.9336</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>0.9380</bold></td>
<td valign="top" align="center"><bold>0.9738</bold></td>
<td valign="top" align="center"><bold>0.9378</bold></td>
<td valign="top" align="center"><bold>0.9014</bold></td>
<td valign="top" align="center"><bold>0.9376</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Bold values indicate optimal values</italic>.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.5.2. Comparative Experiment for Two Versions of Skip-Block</title>
<p>In order to further improve our n-ClsNet classification network ACC of thyroid nodules, we consider that the batch normalization layer possesses the advantage of reducing the risk of overfitting and mitigating the disappearance of gradients. We also tried adding a batch normalization layer in the skip connection layer to improve network performance. Experiments show that this attempt is successful. In comparison with the &#x0201C;No-Batch-Normalization,&#x0201D; the <italic>ACC</italic> increases from 0.9341 to 0.938, the <italic>AP</italic> increases from 0.9726 to 0.9738, the <italic>F</italic>1&#x02212;<italic>score</italic> increases from 0.9337 to 0.9378, the <italic>Recall</italic> increases from 0.9336 to 0.9376, and the <italic>Precision</italic> increases from 0.8844 to 0.9014 by 1.7%. The <italic>Precision</italic> score of ours is significantly beyond &#x0201C;No-Batch-Normalization&#x0201D; architecture, which shows that our proposed final version skip-block is beneficial for thyroid nodules classification.</p>
</sec>
</sec>
<sec>
<title>4.6. Comparison of HAC Block With ASPP Block</title>
<p>To verify the superiority of the HAC block compared to the ASPP block, we conducted a comparative experiment. According to the principle of the control variable method, in the experiments, we only replaced the HAC block with the ASPP block. As shown in <xref ref-type="table" rid="T3">Table 3</xref>, when comparing our employed HAC block to the ASPP block, the <italic>ACC</italic> increases from 0.9186 to 0.938, the <italic>Specificity</italic> increases from 0.8593 to 0.8906, the <italic>F</italic>1&#x02212;<italic>score</italic> increases from 0.9182 to 0.9378, the <italic>Precision</italic> increases from 0.8758 to 0.9014, and the <italic>Recall</italic> increases from 0.9181 to 0.9376. From the comparison, the classification ACC of our HAC block is much higher than that of the ASPP block.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Comparison to HAC block and the atrous spatial pyramid pooling (ASPP) block.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>ACC</bold></th>
<th valign="top" align="center"><bold>Specificity</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">ASPP</td>
<td valign="top" align="center">0.9186</td>
<td valign="top" align="center">0.8593</td>
<td valign="top" align="center">0.9182</td>
<td valign="top" align="center">0.8758</td>
<td valign="top" align="center">0.9181</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center"><bold>0.9380</bold></td>
<td valign="top" align="center"><bold>0.8906</bold></td>
<td valign="top" align="center"><bold>0.9378</bold></td>
<td valign="top" align="center"><bold>0.9014</bold></td>
<td valign="top" align="center"><bold>0.9376</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Bold values indicate optimal values</italic>.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>4.7. Ablation Study</title>
<p>To evaluate the utility of the multi-scale classification layer, skip-block, and HAC block in our deep learning model, control variable comparison experiment is shown in <xref ref-type="table" rid="T4">Table 4</xref>. Afterward, we perform the ablation studies using the TNUI-2021 dataset as examples:</p>
<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Ablations study for each component of our n-ClsNet framework on the TNUI-2021 dataset.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center"><bold>ACC</bold></th>
<th valign="top" align="center"><bold>AP</bold></th>
<th valign="top" align="center"><bold>F1-score</bold></th>
<th valign="top" align="center"><bold>Precision</bold></th>
<th valign="top" align="center"><bold>Recall</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SkipBlock&#x0002B;HAC</td>
<td valign="top" align="center">0.8760</td>
<td valign="top" align="center">0.9128</td>
<td valign="top" align="center">0.8738</td>
<td valign="top" align="center">0.8025</td>
<td valign="top" align="center">0.8750</td>
</tr>
<tr>
<td valign="top" align="left">SkipBlock&#x0002B;Multiscale</td>
<td valign="top" align="center">0.9147</td>
<td valign="top" align="center">0.9614</td>
<td valign="top" align="center">0.9141</td>
<td valign="top" align="center">0.8600</td>
<td valign="top" align="center">0.9141</td>
</tr>
<tr>
<td valign="top" align="left">SkipBlock&#x0002B;Multiscale&#x0002B;HAC</td>
<td valign="top" align="center"><bold>0.9380</bold></td>
<td valign="top" align="center"><bold>0.9738</bold></td>
<td valign="top" align="center"><bold>0.9378</bold></td>
<td valign="top" align="center"><bold>0.9014</bold></td>
<td valign="top" align="center"><bold>0.9376</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Bold values indicate optimal values</italic>.</p>
</table-wrap-foot>
</table-wrap>
<sec>
<title>4.7.1. Ablation Study for Employing Multi-Scale Classification Layer</title>
<p>We adopted the multi-scale classification layer to obtain multi-scale feature maps to improve the learning ability. As we can see from the &#x0201C;SkipBlock&#x0002B;HAC,&#x0201D; the evaluation score of <italic>ACC</italic>, <italic>AP</italic>, <italic>F</italic>1&#x02212;<italic>score</italic>, <italic>Precision</italic>, and <italic>Recall</italic> have been significantly improved: the <italic>ACC</italic> is increased by 6.2% from 0.876 to 0.938, the <italic>AP</italic> is increased by 6.1% from 0.9128 to 0.9738, the <italic>F</italic>1&#x02212;<italic>score</italic> is increased by 6.4% from 0.8738 to 0.9378, the <italic>Precision</italic> is increased by 9.8% from 0.8025 to 0.9014, and the <italic>Recall</italic> is increased by 6% from 0.875 to 0.9376, respectively. Therefore, the results demonstrate that the multi-scale classification layer is effective.</p>
</sec>
<sec>
<title>4.7.2. Ablation Study for Adopting the HAC</title>
<p>We employed the hybrid atrous convolution substitute downsampling, aiming at increasing receptive field size. As shown in <xref ref-type="table" rid="T4">Table 4</xref>, our selected HAC block improves the <italic>ACC</italic>, <italic>AP</italic>, <italic>F</italic>1&#x02212;<italic>score</italic>, <italic>Precision</italic>, and <italic>Recall</italic> in thyroid nodules classification than &#x0201C;SkipBlock&#x0002B;Multiscale&#x0201D;: the <italic>ACC</italic> increases from 0.9147 to 0.938, the <italic>AP</italic> increases from 0.9614 to 0.9738 by 2.3%, the <italic>F</italic>1&#x02212;<italic>score</italic> increases from 0.9141 to 0.9378 by 2.3%, the <italic>Precision</italic> increases from 0.86 to 0.9014 by 4.1%, and the <italic>Recall</italic> increases from 0.9141 to 0.9376 by 2.3%. Even though the evaluation score of <italic>AP</italic> is already performed very well, it has also improved. It demonstrates that our HAC block is useful for the classification task.</p>
</sec>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>5. Conclusion</title>
<p>We present a multi-scale deep learning model, namely n-ClsNet, to classify benign and malignant thyroid nodules on ultrasound images, which use multi-scale ultrasound images as input. On the one hand, our skip-block adopts the strategy of approximate jump connection to excavate the feature of the thyroid nodule image. On the other hand, we propose the HAC that takes the place of downpooling to increase receptive field size and decrease the spatial resolution of the intermediate feature map. The experimental results demonstrate that the n-ClsNet model can effectively improve the performance of classification in thyroid nodules. Moreover, our method surpass the performance of representative state-of-the-art classification methods in the thyroid nodules classification task.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data Availability Statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>LW, XZ, XN, XL, JL, HZ, QG, MD, TT, EX, and MZ: concept and design. LW, XZ, XN, XL, EX, SC, CC, QG, TT, and MZ: acquisition of data. LW, XZ, XN, XL, QG, and TT: model design. LW, XZ, XN, XL, TT, QG, and MZ: data analysis. LW, XZ, XN, XL, EX, TT, QG, and MZ: manuscript drafting. LW, XZ, XN, XL, JL, HZ, SC, CC, QG, MD, TT, EX, and MZ: approval. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>This work was supported by the National Natural Science Foundation of China under grant nos. 61901120 and 62171133, sponsored by Fujian provincial health technology project (2019-1-33), in part by the Science and Technology Innovation Joint Fund Program of Fujian Province of China under grant no. 2019Y9104.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back><sec sec-type="supplementary-material" id="s10">
<title>Supplementary Material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fnins.2022.878718/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fnins.2022.878718/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Table_1.XLSX" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Acharya</surname> <given-names>U. R.</given-names></name> <name><surname>Faust</surname> <given-names>O.</given-names></name> <name><surname>Sree</surname> <given-names>S. V.</given-names></name> <name><surname>Molinari</surname> <given-names>F.</given-names></name> <name><surname>Garberoglio</surname> <given-names>R.</given-names></name> <name><surname>Suri</surname> <given-names>J.</given-names></name></person-group> (<year>2011</year>). <article-title>Cost-effective and non-invasive automated benign and malignant thyroid lesion classification in 3d contrast-enhanced ultrasound using combination of wavelets and textures: a class of thyroscan algorithms</article-title>. <source>Technol. Cancer Res. Treat</source>. <volume>10</volume>, <fpage>371</fpage>&#x02013;<lpage>380</lpage>. <pub-id pub-id-type="doi">10.7785/tcrt.2012.500214</pub-id><pub-id pub-id-type="pmid">21728394</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Acharya</surname> <given-names>U. R.</given-names></name> <name><surname>Sree</surname> <given-names>S. V.</given-names></name> <name><surname>Krishnan</surname> <given-names>M. M. R.</given-names></name> <name><surname>Molinari</surname> <given-names>F.</given-names></name> <name><surname>Garberoglio</surname> <given-names>R.</given-names></name> <name><surname>Suri</surname> <given-names>J. S.</given-names></name></person-group> (<year>2012</year>). <article-title>Non-invasive automated 3d thyroid lesion classification in ultrasound: a class of thyroscan systems</article-title>. <source>Ultrasonics</source> <volume>52</volume>, <fpage>508</fpage>&#x02013;<lpage>520</lpage>. <pub-id pub-id-type="doi">10.1016/j.ultras.2011.11.003</pub-id><pub-id pub-id-type="pmid">22154208</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Acharya</surname> <given-names>U. R.</given-names></name> <name><surname>Sree</surname> <given-names>S. V.</given-names></name> <name><surname>Krishnan</surname> <given-names>M. M. R.</given-names></name> <name><surname>Molinari</surname> <given-names>F.</given-names></name> <name><surname>Ziele&#x000DD;nik</surname> <given-names>W.</given-names></name> <name><surname>Bardales</surname> <given-names>R. H.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Computer-aided diagnostic system for detection of hashimoto thyroiditis on ultrasound images from a polish population</article-title>. <source>J. Ultrasound Med</source>. <volume>33</volume>, <fpage>245</fpage>&#x02013;<lpage>253</lpage>. <pub-id pub-id-type="doi">10.7863/ultra.33.2.245</pub-id><pub-id pub-id-type="pmid">24449727</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ardakani</surname> <given-names>A. A.</given-names></name> <name><surname>Mohammadzadeh</surname> <given-names>A.</given-names></name> <name><surname>Yaghoubi</surname> <given-names>N.</given-names></name> <name><surname>Ghaemmaghami</surname> <given-names>Z.</given-names></name> <name><surname>Reiazi</surname> <given-names>R.</given-names></name> <name><surname>Jafari</surname> <given-names>A. H.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Predictive quantitative sonographic features on classification of hot and cold thyroid nodules</article-title>. <source>Eur. J. Radiol</source>. <volume>101</volume>, <fpage>170</fpage>&#x02013;<lpage>177</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejrad.2018.02.010</pub-id><pub-id pub-id-type="pmid">29571793</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>C.-Y.</given-names></name> <name><surname>Chen</surname> <given-names>S.-J.</given-names></name> <name><surname>Tsai</surname> <given-names>M.-F.</given-names></name></person-group> (<year>2010</year>). <article-title>Application of support-vector-machine-based method for feature selection and classification of thyroid nodules in ultrasound images</article-title>. <source>Pattern. Recognit</source>. <volume>43</volume>, <fpage>3494</fpage>&#x02013;<lpage>3506</lpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2010.04.023</pub-id></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>Y.</given-names></name> <name><surname>Paul</surname> <given-names>A. K.</given-names></name> <name><surname>Kim</surname> <given-names>N.</given-names></name> <name><surname>Baek</surname> <given-names>J. H.</given-names></name> <name><surname>Choi</surname> <given-names>Y. J.</given-names></name> <name><surname>Ha</surname> <given-names>E. J.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Computer-aided diagnosis for classifying benign versus malignant thyroid nodules based on ultrasound images: a comparison with radiologist-based assessments</article-title>. <source>Med. Phys</source>. <volume>43</volume>, <fpage>554</fpage>&#x02013;<lpage>567</lpage>. <pub-id pub-id-type="doi">10.1118/1.4939060</pub-id><pub-id pub-id-type="pmid">26745948</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>J.</given-names></name> <name><surname>Cheng</surname> <given-names>H.</given-names></name> <name><surname>Ning</surname> <given-names>C.</given-names></name> <name><surname>Huang</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name></person-group> (<year>2011</year>). <article-title>Quantitative measurement for thyroid cancer characterization based on elastography</article-title>. <source>J. Ultrasound Med</source>. <volume>30</volume>, <fpage>1259</fpage>&#x02013;<lpage>1266</lpage>. <pub-id pub-id-type="doi">10.7863/jum.2011.30.9.1259</pub-id><pub-id pub-id-type="pmid">21876097</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fu</surname> <given-names>H.</given-names></name> <name><surname>Cheng</surname> <given-names>J.</given-names></name> <name><surname>Xu</surname> <given-names>Y.</given-names></name> <name><surname>Wong</surname> <given-names>D. W. K.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Cao</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>Joint optic disc and cup segmentation based on multi-label deep network and polar transformation</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>37</volume>, <fpage>1597</fpage>&#x02013;<lpage>1605</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2018.2791488</pub-id><pub-id pub-id-type="pmid">29969410</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gu</surname> <given-names>Z.</given-names></name> <name><surname>Cheng</surname> <given-names>J.</given-names></name> <name><surname>Fu</surname> <given-names>H.</given-names></name> <name><surname>Zhou</surname> <given-names>K.</given-names></name> <name><surname>Hao</surname> <given-names>H.</given-names></name> <name><surname>Zhao</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Ce-net: Context encoder network for 2d medical image segmentation</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>38</volume>, <fpage>2281</fpage>&#x02013;<lpage>2292</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2019.2903562</pub-id><pub-id pub-id-type="pmid">30843824</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gulame</surname> <given-names>M. B.</given-names></name> <name><surname>Dixit</surname> <given-names>V. V.</given-names></name> <name><surname>Suresh</surname> <given-names>M.</given-names></name></person-group> (<year>2021</year>). <article-title>Thyroid nodules segmentation methods in clinical ultrasound images: a review</article-title>. <source>Mater. Today</source>. <volume>45</volume>, <fpage>2270</fpage>&#x02013;<lpage>2276</lpage>. <pub-id pub-id-type="doi">10.1016/j.matpr.2020.10.259</pub-id></citation>
</ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Haugen</surname> <given-names>B. R.</given-names></name> <name><surname>Alexander</surname> <given-names>E. K.</given-names></name> <name><surname>Bible</surname> <given-names>K. C.</given-names></name> <name><surname>Doherty</surname> <given-names>G. M.</given-names></name> <name><surname>Mandel</surname> <given-names>S. J.</given-names></name> <name><surname>Nikiforov</surname> <given-names>Y. E.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>2015 american thyroid association management guidelines for adult patients with thyroid nodules and differentiated thyroid cancer: the american thyroid association guidelines task force on thyroid nodules and differentiated thyroid cancer</article-title>. <source>Thyroid</source> <volume>26</volume>, <fpage>1</fpage>&#x02013;<lpage>133</lpage>. <pub-id pub-id-type="doi">10.1089/thy.2015.0020</pub-id><pub-id pub-id-type="pmid">26462967</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>&#x0201C;Deep residual learning for image recognition,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>770</fpage>&#x02013;<lpage>778</lpage>.<pub-id pub-id-type="pmid">32166560</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Howard</surname> <given-names>A. G.</given-names></name> <name><surname>Zhu</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>B.</given-names></name> <name><surname>Kalenichenko</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>W.</given-names></name> <name><surname>Weyand</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Mobilenets: efficient convolutional neural networks for mobile vision applications</article-title>. <source>arXiv preprint arXiv:1704.04861</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1704.04861</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>G.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Van Der Maaten</surname> <given-names>L.</given-names></name> <name><surname>Weinberger</surname> <given-names>K. Q.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Densely connected convolutional networks,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Honolulu, HI</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>4700</fpage>&#x02013;<lpage>4708</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Iakovidis</surname> <given-names>D. K.</given-names></name> <name><surname>Keramidas</surname> <given-names>E. G.</given-names></name> <name><surname>Maroulis</surname> <given-names>D.</given-names></name></person-group> (<year>2010</year>). <article-title>Fusion of fuzzy statistical distributions for classification of thyroid ultrasound patterns</article-title>. <source>Artif. Intell. Med</source>. <volume>50</volume>, <fpage>33</fpage>&#x02013;<lpage>41</lpage>. <pub-id pub-id-type="doi">10.1016/j.artmed.2010.04.004</pub-id><pub-id pub-id-type="pmid">20427164</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Iandola</surname> <given-names>F. N.</given-names></name> <name><surname>Han</surname> <given-names>S.</given-names></name> <name><surname>Moskewicz</surname> <given-names>M. W.</given-names></name> <name><surname>Ashraf</surname> <given-names>K.</given-names></name> <name><surname>Dally</surname> <given-names>W. J.</given-names></name> <name><surname>Keutzer</surname> <given-names>K.</given-names></name></person-group> (<year>2016</year>). <article-title>Squeezenet: alexnet-level accuracy with 50x fewer parameters and &#x0003C;0.5 mb model size</article-title>. <source>arXiv preprint arXiv:1602.07360</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1602.07360</pub-id></citation>
</ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krizhevsky</surname> <given-names>A.</given-names></name> <name><surname>Sutskever</surname> <given-names>I.</given-names></name> <name><surname>Hinton</surname> <given-names>G. E.</given-names></name></person-group> (<year>2012</year>). <article-title>Imagenet classification with deep convolutional neural networks</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>25</volume>, <fpage>1097</fpage>&#x02013;<lpage>1105</lpage>. <pub-id pub-id-type="doi">10.1145/3065386</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kumari</surname> <given-names>S. V.</given-names></name> <name><surname>Rani</surname> <given-names>K. U.</given-names></name></person-group> (<year>2019</year>). <article-title>&#x0201C;Analysis on various feature extraction methods for medical image classification,&#x0201D;</article-title> in <source>International Conference On Computational and Bio Engineering</source> (<publisher-loc>Tirupati</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>19</fpage>&#x02013;<lpage>31</lpage>.<pub-id pub-id-type="pmid">9241522</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>T.</given-names></name> <name><surname>Guo</surname> <given-names>Q.</given-names></name> <name><surname>Lian</surname> <given-names>C.</given-names></name> <name><surname>Ren</surname> <given-names>X.</given-names></name> <name><surname>Liang</surname> <given-names>S.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Automated detection and classification of thyroid nodules in ultrasound images using clinical-knowledge-guided convolutional neural networks</article-title>. <source>Med. Image Anal</source>. <volume>58</volume>, <fpage>101555</fpage>. <pub-id pub-id-type="doi">10.1016/j.media.2019.101555</pub-id><pub-id pub-id-type="pmid">31520984</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Cheng</surname> <given-names>M.-M.</given-names></name> <name><surname>Hu</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>K.</given-names></name> <name><surname>Bai</surname> <given-names>X.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Richer convolutional features for edge detection,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Venice</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>3000</fpage>&#x02013;<lpage>3009</lpage>.<pub-id pub-id-type="pmid">30387723</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lu</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>T.</given-names></name> <name><surname>Shi</surname> <given-names>L.</given-names></name> <name><surname>Qiu</surname> <given-names>J.</given-names></name></person-group> (<year>2020</year>). <article-title>Classification of ti-rads class-4 thyroid nodules via ultrasound-based radiomics and multi-kernel learning</article-title>. <source>Int. J. Radiat. Oncol. Biol. Phys</source>. <volume>108</volume>, <fpage>e848</fpage>. <pub-id pub-id-type="doi">10.1016/j.ijrobp.2020.07.401</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Miranda-Filho</surname> <given-names>A.</given-names></name> <name><surname>Lortet-Tieulent</surname> <given-names>J.</given-names></name> <name><surname>Bray</surname> <given-names>F.</given-names></name> <name><surname>Cao</surname> <given-names>B.</given-names></name> <name><surname>Franceschi</surname> <given-names>S.</given-names></name> <name><surname>Vaccarella</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Thyroid cancer incidence trends by histology in 25 countries: a population-based study</article-title>. <source>Lancet Diabetes Endocrinol</source>. <volume>9</volume>, <fpage>225</fpage>&#x02013;<lpage>234</lpage>. <pub-id pub-id-type="doi">10.1016/S2213-8587(21)00027-9</pub-id><pub-id pub-id-type="pmid">33662333</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ouyang</surname> <given-names>F.-,s.</given-names></name> <name><surname>Guo</surname> <given-names>B.-,l.</given-names></name> <name><surname>Ouyang</surname> <given-names>L.-,z.</given-names></name> <name><surname>Liu</surname> <given-names>Z.-,w.</given-names></name> <name><surname>Lin</surname> <given-names>S.-,j.</given-names></name> <name><surname>Meng</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Comparison between linear and nonlinear machine-learning algorithms for the classification of thyroid nodules</article-title>. <source>Eur. J. Radiol</source>. <volume>113</volume>, <fpage>251</fpage>&#x02013;<lpage>257</lpage>. <pub-id pub-id-type="doi">10.1016/j.ejrad.2019.02.029</pub-id><pub-id pub-id-type="pmid">30927956</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prochazka</surname> <given-names>A.</given-names></name> <name><surname>Gulati</surname> <given-names>S.</given-names></name> <name><surname>Holinka</surname> <given-names>S.</given-names></name> <name><surname>Smutek</surname> <given-names>D.</given-names></name></person-group> (<year>2019a</year>). <article-title>Classification of thyroid nodules in ultrasound images using direction-independent features extracted by two-threshold binary decomposition</article-title>. <source>Technol. Cancer Res. Treat</source>. <volume>18</volume>, 1533033819830748. <pub-id pub-id-type="doi">10.1177/1533033819830748</pub-id><pub-id pub-id-type="pmid">30774015</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Prochazka</surname> <given-names>A.</given-names></name> <name><surname>Gulati</surname> <given-names>S.</given-names></name> <name><surname>Holinka</surname> <given-names>S.</given-names></name> <name><surname>Smutek</surname> <given-names>D.</given-names></name></person-group> (<year>2019b</year>). <article-title>Patch-based classification of thyroid nodules in ultrasound images using direction independent features extracted by two-threshold binary decomposition</article-title>. <source>Comput. Med. Imaging Graphics</source> <volume>71</volume>, <fpage>9</fpage>&#x02013;<lpage>18</lpage>. <pub-id pub-id-type="doi">10.1016/j.compmedimag.2018.10.001</pub-id><pub-id pub-id-type="pmid">30453231</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Raghavendra</surname> <given-names>U.</given-names></name> <name><surname>Acharya</surname> <given-names>U. R.</given-names></name> <name><surname>Gudigar</surname> <given-names>A.</given-names></name> <name><surname>Tan</surname> <given-names>J. H.</given-names></name> <name><surname>Fujita</surname> <given-names>H.</given-names></name> <name><surname>Hagiwara</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Fusion of spatial gray level dependency and fractal texture features for the characterization of thyroid lesions</article-title>. <source>Ultrasonics</source> <volume>77</volume>, <fpage>110</fpage>&#x02013;<lpage>120</lpage>. <pub-id pub-id-type="doi">10.1016/j.ultras.2017.02.003</pub-id><pub-id pub-id-type="pmid">28219805</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sharifi</surname> <given-names>Y.</given-names></name> <name><surname>Bakhshali</surname> <given-names>M. A.</given-names></name> <name><surname>Dehghani</surname> <given-names>T.</given-names></name> <name><surname>DanaiAshgzari</surname> <given-names>M.</given-names></name> <name><surname>Sargolzaei</surname> <given-names>M.</given-names></name> <name><surname>Eslami</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Deep learning on ultrasound images of thyroid nodules</article-title>. <source>Biocybernetics Biomed. Eng</source>. <volume>41</volume>, <fpage>636</fpage>&#x02013;<lpage>655</lpage>. <pub-id pub-id-type="doi">10.1016/j.bbe.2021.02.008</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Simonyan</surname> <given-names>K.</given-names></name> <name><surname>Zisserman</surname> <given-names>A.</given-names></name></person-group> (<year>2014</year>). <article-title>Very deep convolutional networks for large-scale image recognition</article-title>. <source>arXiv preprint arXiv:1409.1556</source>. <pub-id pub-id-type="doi">10.48550/arXiv.1409.1556</pub-id></citation>
</ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Song</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Qin</surname> <given-names>H.</given-names></name> <name><surname>Zhang</surname> <given-names>B.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Multitask cascade convolution neural networks for automatic thyroid nodule detection and recognition</article-title>. <source>IEEE J. Biomed. Health Inform</source>. <volume>23</volume>, <fpage>1215</fpage>&#x02013;<lpage>1224</lpage>. <pub-id pub-id-type="doi">10.1109/JBHI.2018.2852718</pub-id><pub-id pub-id-type="pmid">29994412</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sun</surname> <given-names>J.</given-names></name> <name><surname>Sun</surname> <given-names>T.</given-names></name> <name><surname>Yuan</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Shi</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>Y.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Automatic diagnosis of thyroid ultrasound image based on fcn-alexnet and transfer learning,&#x0201D;</article-title> in <source>2018 IEEE 23rd International Conference on Digital Signal Processing (DSP)</source> (<publisher-loc>Shanghai</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Jia</surname> <given-names>Y.</given-names></name> <name><surname>Sermanet</surname> <given-names>P.</given-names></name> <name><surname>Reed</surname> <given-names>S.</given-names></name> <name><surname>Anguelov</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>&#x0201C;Going deeper with convolutions,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Boston, MA</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>9</lpage>.</citation>
</ref>
<ref id="B32">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>M.</given-names></name> <name><surname>Yuan</surname> <given-names>C.</given-names></name> <name><surname>Wu</surname> <given-names>D.</given-names></name> <name><surname>Zeng</surname> <given-names>Y.</given-names></name> <name><surname>Zhong</surname> <given-names>S.</given-names></name> <name><surname>Qiu</surname> <given-names>W.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Automatic segmentation and classification of thyroid nodules in ultrasound images with convolutional neural networks,&#x0201D;</article-title> in <source>Proceedings of the 23rd International Conference on Medical Image Computing and Computer-Assisted Intervention</source> (<publisher-loc>Lima</publisher-loc>: <publisher-name>Springer</publisher-name>), <fpage>109</fpage>&#x02013;<lpage>115</lpage>.<pub-id pub-id-type="pmid">28762196</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>P.</given-names></name> <name><surname>Chen</surname> <given-names>P.</given-names></name> <name><surname>Yuan</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>D.</given-names></name> <name><surname>Huang</surname> <given-names>Z.</given-names></name> <name><surname>Hou</surname> <given-names>X.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>&#x0201C;Understanding convolution for semantic segmentation,&#x0201D;</article-title> in <source>2018 IEEE Winter Conference on Applications of Computer Vision (WACV)</source> (<publisher-loc>Lake Tahoe, NV</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1451</fpage>&#x02013;<lpage>1460</lpage>.</citation>
</ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xiong</surname> <given-names>W.</given-names></name> <name><surname>Jia</surname> <given-names>X.</given-names></name> <name><surname>Yang</surname> <given-names>D.</given-names></name> <name><surname>Ai</surname> <given-names>M.</given-names></name> <name><surname>Li</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>S.</given-names></name></person-group> (<year>2021</year>). <article-title>Dp-linknet: A convolutional network for historical document image binarization</article-title>. <source>KSII Trans. Internet Inf. Syst</source>. <volume>15</volume>, <fpage>1778</fpage>&#x02013;<lpage>1797</lpage>. <pub-id pub-id-type="doi">10.3837/tiis.2021.05.011</pub-id></citation>
</ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname> <given-names>W.</given-names></name> <name><surname>Dong</surname> <given-names>Y.</given-names></name> <name><surname>Du</surname> <given-names>Q.</given-names></name> <name><surname>Qiang</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>K.</given-names></name> <name><surname>Zhao</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Integrate domain knowledge in training multi-task cascade deep learning model for benign-malignant thyroid nodule classification on ultrasound images</article-title>. <source>Eng. Appl. Artif. Intell</source>. <volume>98</volume>, <fpage>104064</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2020.104064</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Zhao</surname> <given-names>C.</given-names></name> <name><surname>Guo</surname> <given-names>L.</given-names></name> <name><surname>Li</surname> <given-names>X.</given-names></name> <name><surname>Luo</surname> <given-names>Y.</given-names></name> <name><surname>Lu</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2019a</year>). <article-title>&#x0201C;Diagnosis of thyroid nodules in ultrasound images using two combined classification modules,&#x0201D;</article-title> in <source>2019 12th International Congress on Image and Signal Processing, BioMedical Engineering and Informatics (CISP-BMEI)</source> (<publisher-loc>Suzhou</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1</fpage>&#x02013;<lpage>5</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J.</given-names></name> <name><surname>Xie</surname> <given-names>Y.</given-names></name> <name><surname>Xia</surname> <given-names>Y.</given-names></name> <name><surname>Shen</surname> <given-names>C.</given-names></name></person-group> (<year>2019b</year>). <article-title>Attention residual learning for skin lesion classification</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>38</volume>, <fpage>2092</fpage>&#x02013;<lpage>2103</lpage>. <pub-id pub-id-type="doi">10.1109/TMI.2019.2893944</pub-id><pub-id pub-id-type="pmid">33321864</pub-id></citation></ref>
</ref-list> 
</back>
</article>