<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<?covid-19-tdm?>
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Oncol.</journal-id>
<journal-title>Frontiers in Oncology</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Oncol.</abbrev-journal-title>
<issn pub-type="epub">2234-943X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fonc.2021.781798</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Oncology</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Automatic Sequence-Based Network for Lung Diseases Detection in Chest CT</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Hao</surname>
<given-names>Jinkui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1489080"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Xie</surname>
<given-names>Jianyang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1408372"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Ri</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hao</surname>
<given-names>Huaying</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Yuhui</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1409473"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yan</surname>
<given-names>Kun</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Ruirui</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zheng</surname>
<given-names>Yalin</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1316816"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zheng</surname>
<given-names>Jianjun</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Liu</surname>
<given-names>Jiang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhang</surname>
<given-names>Jingfeng</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Zhao</surname>
<given-names>Yitian</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<xref ref-type="aff" rid="aff8">
<sup>8</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/1362172"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Cixi Institute of Biomedical Engineering, Ningbo Institute of Material Technology and Engineering, Chinese Academy of Sciences</institution>, <addr-line>Ningbo</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>School of Optical Technology, University of Chinese Academy of Sciences</institution>, <addr-line>Beijing</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Hwa Mei Hospital, University of Chinese Academy of Sciences</institution>, <addr-line>Ningbo</addr-line>, <country>China</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>School of Medicine, Ningbo University</institution>, <addr-line>Ningbo</addr-line>, <country>China</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Department of Eye and Vision Science, University of Liverpool</institution>, <addr-line>Liverpool</addr-line>, <country>United Kingdom</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Department of Computer Science and Engineering, Southern University of Science and Technology</institution>, <addr-line>Shenzhen</addr-line>, <country>China</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Zhejiang International Scientific and Technological Cooperative Base of Biomedical Materials and Technology, Ningbo Institute of Material Technology and Engineering, Chinese Academy of Sciences</institution>, <addr-line>Ningbo</addr-line>, <country>China</country>
</aff>
<aff id="aff8">
<sup>8</sup>
<institution>Zhejiang Engineering Research Center for Biomedical Materials, Ningbo Institute of Material Technology and Engineering, Chinese Academy of Sciences</institution>, <addr-line>Ningbo</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Guang Yang, Imperial College London, United Kingdom</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Anand Nayyar, Duy Tan University, Vietnam; Seyedali Mirjalili, Torrens University Australia, Australia; Tao Zhou, Nanjing University of Science and Technology, China; Zhili Chen, Shenyang Jianzhu University, China</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Yitian Zhao, <email xlink:href="mailto:yitian.zhao@nimte.ac.cn">yitian.zhao@nimte.ac.cn</email>; Jingfeng Zhang, <email xlink:href="mailto:jingfeng.zhang@163.com">jingfeng.zhang@163.com</email>
</p>
</fn>
<fn fn-type="equal" id="fn003">
<p>&#x2020;These authors have contributed equally to this work</p>
</fn>
<fn fn-type="other" id="fn002">
<p>This article was submitted to Cancer Imaging and Image-directed Interventions, a section of the journal Frontiers in Oncology</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>02</day>
<month>12</month>
<year>2021</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>11</volume>
<elocation-id>781798</elocation-id>
<history>
<date date-type="received">
<day>23</day>
<month>09</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>01</day>
<month>11</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2021 Hao, Xie, Liu, Hao, Ma, Yan, Liu, Zheng, Zheng, Liu, Zhang and Zhao</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Hao, Xie, Liu, Hao, Ma, Yan, Liu, Zheng, Zheng, Liu, Zhang and Zhao</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Objective</title>
<p>To develop an accurate and rapid computed tomography (CT)-based interpretable AI system for the diagnosis of lung diseases.</p>
</sec>
<sec>
<title>Background</title>
<p>Most existing AI systems only focus on viral pneumonia (e.g., COVID-19), specifically, ignoring other similar lung diseases: e.g., bacterial pneumonia (BP), which should also be detected during CT screening. In this paper, we propose a unified sequence-based pneumonia classification network, called SLP-Net, which utilizes consecutiveness information for the differential diagnosis of viral pneumonia (VP), BP, and normal control cases from chest CT volumes.</p>
</sec>
<sec>
<title>Methods</title>
<p>Considering consecutive images of a CT volume as a time sequence input, compared with previous 2D slice-based or 3D volume-based methods, our SLP-Net can effectively use the spatial information and does not need a large amount of training data to avoid overfitting. Specifically, sequential convolutional neural networks (CNNs) with multi-scale receptive fields are first utilized to extract a set of higher-level representations, which are then fed into a convolutional long short-term memory (ConvLSTM) module to construct axial dimensional feature maps. A novel adaptive-weighted cross-entropy loss (ACE) is introduced to optimize the output of the SLP-Net with a view to ensuring that as many valid features from the previous images as possible are encoded into the later CT image. In addition, we employ sequence attention maps for auxiliary classification to enhance the confidence level of the results and produce a case-level prediction.</p>
</sec>
<sec>
<title>Results</title>
<p>For evaluation, we constructed a dataset of 258 chest CT volumes with 153 VP, 42 BP, and 63 normal control cases, for a total of 43,421 slices. We implemented a comprehensive comparison between our SLP-Net and several state-of-the-art methods across the dataset. Our proposed method obtained significant performance without a large amount of data, outperformed other slice-based and volume-based approaches. The superior evaluation performance achieved in the classification experiments demonstrated the ability of our model in the differential diagnosis of VP, BP and normal cases.</p>
</sec>
</abstract>
<kwd-group>
<kwd>deep learning</kwd>
<kwd>CT</kwd>
<kwd>CNN</kwd>
<kwd>ConvLSTM</kwd>
<kwd>lung diseases</kwd>
</kwd-group>
<counts>
<fig-count count="8"/>
<table-count count="6"/>
<equation-count count="8"/>
<ref-count count="52"/>
<page-count count="14"/>
<word-count count="7971"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1">
<title>1 Introduction</title>
<p>COVID-19, the latest in viral pneumonia diseases, is an acute respiratory syndrome that has spread rapidly around the world since the end of 2019, having a devastating effect on the health and well-being of the global population (<xref ref-type="bibr" rid="B1">1</xref>, <xref ref-type="bibr" rid="B2">2</xref>). To diagnose viral pneumonia (limited to COVID-19 in our work), reverse transcription-polymerase chain reaction (RT-PCR) has widely been accepted as the gold standard. However, shortages of equipment and strict requirements for testing environments limit the rapid and accurate screening of suspected subjects. Furthermore, RT-PCR testing is also reported to suffer from a high false-negative rate (<xref ref-type="bibr" rid="B3">3</xref>), with a low sensitivity of only 71%. In clinical practice, radiological imaging techniques, e.g., X-rays and computed tomography (CT), have also been demonstrated to be effective in diagnosis, and also follow-up assessment and evaluation of disease evolution (<xref ref-type="bibr" rid="B4">4</xref>, <xref ref-type="bibr" rid="B5">5</xref>). CT is the most widely used imaging technique, due to its high resolution and three-dimensional (3D) view, and its relatively high detection sensitivity of around 98% (<xref ref-type="bibr" rid="B6">6</xref>). For example, the study (<xref ref-type="bibr" rid="B5">5</xref>) found that the dynamic lesion process of viral pneumonia (from ground-glass opacity in the early stage to pulmonary consolidation in the late stage) can be observed in CT scans, and its CT manifestations have been emphasized.</p>
<p>Bacterial and viral pathogens are the two leading causes of pneumonia, but require very different forms of management (<xref ref-type="bibr" rid="B7">7</xref>). Bacterial pneumonia requires urgent referral for immediate antibiotic treatment, while viral pneumonia is treated with supportive care. Therefore, accurate classification of different types of pneumonia is imperative for timely diagnosis and treatment. However, the imaging features of viral and bacterial infections are not often compared, and the only imaging feature that was significantly different between the viral and bacterial lung infection was the frequency of diffuse airspace disease (<xref ref-type="bibr" rid="B8">8</xref>). In the case of a typical viral pneumonia, in clinical practice, it is difficult to accurately differentiate viral pneumonia from bacterial pneumonia. See <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> as an example. In clinical practice, especially in primary medical institutions, the consistency of imaging diagnosis of pneumonia pathogens is poor (<xref ref-type="bibr" rid="B9">9</xref>&#x2013;<xref ref-type="bibr" rid="B11">11</xref>). Moreover, it is time consuming for radiologists to read CT volumes that contain hundreds of 2D slices. As such, it is of great practical significance to quickly and accurately identify pathogens to guide individualized anti-infectious treatment and minimize and delay the occurrence of drug resistance.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Example axial CT slices of viral pneumonia <bold>(A)</bold>, bacterial pneumonia <bold>(B)</bold>. Accurate classification of different types of pneumonia is imperative for timely diagnosis and treatment. However, viral pneumonia and bacterial pneumonia display similar appearances in a CT image, which makes it difficult to accurately differentiate a patient with viral pneumonia from a case of bacterial pneumonia.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-11-781798-g001.tif"/>
</fig>
<p>As an emerging technology in medical image analysis, artificial intelligence (AI) has been widely employed for lesion segmentation, and for clinical assessment and diagnosis of lung-related diseases <italic>via</italic> radiological imaging (<xref ref-type="bibr" rid="B12">12</xref>, <xref ref-type="bibr" rid="B13">13</xref>). Recently, many novel AI techniques for viral pneumonia have been presented (<xref ref-type="bibr" rid="B1">1</xref>). For instance, Ouyang et&#xa0;al. (<xref ref-type="bibr" rid="B14">14</xref>) proposed a dual-sampling attention network for the differential diagnosis of COVID-19 from Community Acquired Pneumonia (CAP) (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B15">15</xref>), and Fan et&#xa0;al. (<xref ref-type="bibr" rid="B16">16</xref>) introduced an automatic COVID-19 lung infection lesion segmentation method using a deep network. These works are useful for detecting and controlling of the spread of COVID-19. However, there are very few studies on differentiating COVID-19 from other etiological pneumonias, despite success in using deep learning (DL) approaches to discriminate bacterial and viral pneumonias in pediatric chest radiographs (<xref ref-type="bibr" rid="B17">17</xref>, <xref ref-type="bibr" rid="B18">18</xref>).<sup>
<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref>
</sup>
</p>
<p>Further, most existing works make use of 2D CT slices, and the lack of continuity information makes it impossible to capture the true spatial distribution of the lesion in the lungs. To this end, some recent studies have attempted to use entire 3D volumes to train a 3D classification or segmentation model directly (<xref ref-type="bibr" rid="B14">14</xref>, <xref ref-type="bibr" rid="B19">19</xref>) ,achieving a slightly better performance than the approaches based on 2D slices. However, these 3D volume based approaches greatly increase the computational load, and require much more powerful and expensive hardware configurations. Additionally, 3D volumes may contain large portions of redundant information, which leads to great difficulty in accurately identifying small lesions. The imaging appearance of viral and bacterial lung infection has considerable similarity, and that, in any individual case, the viral pneumonia cannot reliably be distinguished from bacterial infections (<xref ref-type="bibr" rid="B8">8</xref>, <xref ref-type="bibr" rid="B20">20</xref>). For example, viral pneumonia and bacterial pneumonia have some image features in common, such as ground glass opacities and interstitial changes in the peripheral zone of lungs, and accompanied by partial consolidation. The only imaging feature that was significantly different between the viral and bacterial lung infection was the frequency of diffuse airspace disease (<xref ref-type="bibr" rid="B8">8</xref>). Precise characterization of the spatial morphology of the infected regional lesions is essential to distinguish the two infection types by CT imaging.</p>
<p>In this paper, we treat the spatially continuous 2D CT slices as a time sequence and proposed a unified sequence-based pneumonia classification network (SLP-Net) for differentiating viral pneumonia (VP) from bacterial pneumonia (BP) and normal control cases. Our network comprises a CNN encoder and the ConvLSTM module, and sequence attention maps are used for auxiliary classification. As stated above, the precise characterization of the lesion is the key to distinguish the different pneumonia types. The combination of these components ensures the model pay more attention to spatial morphology of the lesion during the decision making. Specifically, the encoder with multi-scale receptive fields is first used to extract local representations of the sequence. Then we apply the ConvLSTM to acquire spatial information of these sequence features, modeling the distribution of the lesion. To optimize the SLP-Net, we introduce a novel adaptive-weighted cross-entropy (ACE) loss, with a view to ensuring that as many valid features from the previous images as possible are encoded into the subsequent CT image. Given the fact that the final diagnosis conclusion needs to be made for each patient, case-based prediction rather than a slice- or sequence-based prediction is more valuable. To obtain case-based prediction, in addition to the classification result of the sequence, we also use sequence attention maps to aid the case-level classification, aiming to enhance the confidence of the results. We collect a dataset of 258 chest CT volumes (153 VP, 42 BP, and 63 normal control cases). The experimental results show that the proposed SLP-Net achieves an accurate classification performance of viral pneumonia, bacterial pneumonia, and normal control, which could benefit the large-scale screening and control of viral pneumonia, and also enable efficient treatment for different types of pneumonia.</p>
<p>We organize the remainder of this paper as follows. In Section 2, the existing methods of AI-enpowered viral pneumonia analysis are briefly reviewed. In Section 3 we give detailed descriptions of collected datasets. Section 4 introduces the proposed SLP-Net. In Section 5, we present the experimental results and discuss the effectiveness, robustness, and efficiency of the SLP-Net. Section 6 concludes the paper and indicates directions for future work.</p>
</sec>
<sec id="s2">
<title>2 Related Work</title>
<p>AI-based medical image analysis plays an essential role in the global fight against COVID-19, and a considerable number of approaches have been proposed in the past five months. This body of work on COVID-19 has focused primarily on two problems: lesion segmentation (<xref ref-type="bibr" rid="B16">16</xref>, <xref ref-type="bibr" rid="B21">21</xref>, <xref ref-type="bibr" rid="B22">22</xref>), and automated screening (<xref ref-type="bibr" rid="B23">23</xref>&#x2013;<xref ref-type="bibr" rid="B31">31</xref>). For example (<xref ref-type="bibr" rid="B16">16</xref>), recently introduced a parallel partial decoder to aggregate high-level features, using an implicit reverse attention and explicit edge-attention to model boundaries and enhance representations so as to identify infected regions from 2D chest CT slices. To alleviate the shortage of labeled data, a semi-supervised segmentation framework based on a randomly selected propagation strategy was applied by (<xref ref-type="bibr" rid="B21">21</xref>). They proposed a relational approach, in which a non-local neural network module was introduced to efficiently learn both visual and geometric relationships among all convolutional features.</p>
<p>However, automated viral pneumonia (e.g., COVID-19) screening has attracted even more attention. For instance (<xref ref-type="bibr" rid="B32">32</xref>), introduced a COVID-19 detection method with multi-task DL approaches, using an inception residual recurrent convolutional neural network (CNN) with transfer learning. Their detection model achieved 84.67% accuracy from X-ray images (<xref ref-type="bibr" rid="B33">33</xref>). proposed a deep features fusion and ranking technique to detect COVID-19 in its early phase. They employed a pre-trained CNN structure to obtain a set of features, which were subsequently fused and evaluated with a support vector machine (SVM) classifier. In the classification task of COVID-19 and no COVID-19, their proposed method obtained 98.27% accuracy on their own dataset (<xref ref-type="bibr" rid="B34">34</xref>). applied a modified residual network, called DeepPneumonia, based on ResNet50 for slice-level classification, and could discriminate the COVID-19 patients from the bacteria pneumonia patients with an AUC of 0.95 (<xref ref-type="bibr" rid="B23">23</xref>). built multiple deep convolutional neural models for classifying chest X-ray images into normal and COVID-19 cases, which obtained 96.1% accuracy (<xref ref-type="bibr" rid="B35">35</xref>). proposed a unified latent representation to explore multiple features describing CT images from different views, a method that can completely encode information from different features aspects and is endowed with a promising class structure for separability. Performance in diagnosis for COVID-19 and community-acquired pneumonia (CAP) is 95.5% in terms of accuarcy. An infection size-aware random forest method was introduced by (<xref ref-type="bibr" rid="B15">15</xref>) for the differentiation of COVID-19 from CAP, in which patients were automatically categorized into groups with different extensions of infected lesion sizes, followed by generation of random forests with each group for classification. The method achieved an accuracy of 89.4% in discriminating COVID-19 from CAP.</p>
<p>However, all of the above mentioned works are based on 2D images, the spatial correlation between consecutive CT scans is neglected by most slice-based methods, despite this being essential for the screening of lung diseases. A variety of volume-based methods have been proposed in an attempt to address this deficiency (<xref ref-type="bibr" rid="B19">19</xref>). proposed an attention-based deep 3D multi-instance learning method to screen COVID-19 from 3D chest CT sacns, using a weakly supervised learning framework that incorporates an attention mechanism into deep multi-instance learning, achieving an accuracy of 97.9% (<xref ref-type="bibr" rid="B14">14</xref>). proposed a 3D CNN to diagnose COVID-19 from CAP, in which a novel online attention module is combined with a dual-sampling strategy. The online attention module focuses on the infected regions when making diagnostic decisions. The dual-sampling strategy mitigates the imbalanced distribution in the sizes of infected regions between COVID-19 and CAP. Their method was evaluated in private dataset and achieved an accuracy of 87.5%. These 3D volume based approaches greatly increase the computational load, and require much more powerful and expensive hardware configurations. Additionally, 3D volumes may contain large portions of redundant information, which leads to great difficulty in accurately identifying small lesions. Overall, 2D slice based methods cannot take advantage of spatial continuity information and 3D volume based methods require much more expensive hardware configurations. In order to take advantage of the complementary information of 2D slices and 3D volumes, we treat the spatially continuous 2D CT slices as a time sequence, and divide the volume into multiple different temporal sequences of consecutive slices as the input.</p>
</sec>
<sec id="s3" sec-type="materials|methods">
<title>3 Materials and Methods</title>
<sec id="s3_1">
<title>3.1 Materials</title>
<p>A total of 258 subjects were enrolled into this study, with 258 CT volumes, corresponding to 43,421 slices. Of the 258 subjects, 42 patients were confirmed positive for BP by clinical diagnosis (age: 59.5 &#xb1; 27.2; male/female: 36/6), 153 patients were positive for VP, confirmed by RT-PCR (age: 52.3 &#xb1; 12.7; male/female: 68/85), and 63 were control subjects (age: 35.8 &#xb1; 11.7; male/female: 33/30). The CT volumes of normal and VP patients were captured between January 29, 2020 and February 18, 2020, and the BP data was collected between January 2, 2019 and February 19, 2020. There is no statistically significant difference between the ages of the VP and BP subjects (<italic>P</italic> = &gt;0.05), but both groups are significantly older than patients in the normal group (<italic>P</italic> &lt; 0.001). CT examinations of all the enrolled patients were performed on a ScintCare CT16 (Minfound Inc, China) with standard chest imaging protocols. All the patients underwent CT scans during the end-inspiration without the administration of contrast material. Related parameters for chest CT scanning were listed as follows: field of view (FOV), 360 mm; tube voltage, 120 kV; tube current, 240 mA; helical mode; slice thickness, 5 mm; pitch, 1.5; collimation 16 &#xd7; 1.2 mm; gantry rotation speed, 0.5 s/r; matrix, 512 &#xd7; 512; software version, syngo CT 2014A; mediastinal window: window width of 350 HU, with a window level of 40 HU; and lung window: window width of 1,300 HU, with a window level of &#x2212;500 HU.</p>
<p>CT volumes were retrospectively collected according to the history of laboratory investigations (e.g., sputum culture and reverse transcription-polymerase chain reaction), which we can generate the case-level labels. Meanwhile, professional radiologists (from the Hwa Mei Hospital, University of Chinese Academy of Sciences, Ningbo, China.) picked out the slice containing the infected region in each volume for the subsequent automatic generation of sequence labels: each volume was be divided into overlapping sequences containing <italic>n</italic> slices, with <italic>k</italic> overlapping slices between two sequences. If a sequence from VP volume contains the infected slice(s), the label for that sequence is 1; if it is from BP volume and contains infected slice(s), the label is 2; the label from normal volume is 0. It is worth noting that <italic>k</italic> and <italic>n</italic> are both hyperparameters, and in <italic>Sensitivities to Hyperparameters</italic> we discuss how to choose the values of these, as well as their impact on the classification results.</p>
<p>The sequences generated from the volume of VP and BP were not used for training if they did not contain any slices with lesion regions. If all normal slices from patients are used for training instead of excluding them, it will increase the proportion of normal control samples in the training set.The data imbalance may cause the model to over-fit the normal samples and tend to predict the samples with the lesion as normal. To avoid this, we exclude the normal slice from patients. For training and evaluation of the proposed method, as shown in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>, we split 258 volumes into 168 volumes (95 VP, 30 BP, and 43 normal controls) for training and 90 (58 VP, 12 BP, and 20 normal controls) volumes for testing.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Characteristics of training and testing CT dataset for identifying viral pneumonia (VP) from bacterial pneumonia (BP) and normal controls.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" rowspan="2" align="left">Cohort</th>
<th valign="top" colspan="2" align="center">VP</th>
<th valign="top" colspan="2" align="center">BP</th>
<th valign="top" colspan="2" align="center">Normal controls</th>
<th valign="top" colspan="2" align="center">Total</th>
</tr>
<tr>
<th valign="top" align="center">Volumes</th>
<th valign="top" align="center">Slices</th>
<th valign="top" align="center">Volumes</th>
<th valign="top" align="center">Slices</th>
<th valign="top" align="center">Volumes</th>
<th valign="top" align="center">Slices</th>
<th valign="top" align="center">Volumes</th>
<th valign="top" align="center">Slices</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Training set</td>
<td valign="top" align="center">95</td>
<td valign="top" align="center">15,931</td>
<td valign="top" align="center">30</td>
<td valign="top" align="center">5,068</td>
<td valign="top" align="center">43</td>
<td valign="top" align="center">7,310</td>
<td valign="top" align="center">168</td>
<td valign="top" align="center">28,309</td>
</tr>
<tr>
<td valign="top" align="left">Testing set</td>
<td valign="top" align="center">58</td>
<td valign="top" align="center">7,832</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">3,107</td>
<td valign="top" align="center">20</td>
<td valign="top" align="center">4,173</td>
<td valign="top" align="center">90</td>
<td valign="top" align="center">15,112</td>
</tr>
<tr>
<td valign="top" align="left">Total</td>
<td valign="top" align="center">153</td>
<td valign="top" align="center">23,763</td>
<td valign="top" align="center">42</td>
<td valign="top" align="center">8,175</td>
<td valign="top" align="center">63</td>
<td valign="top" align="center">11,483</td>
<td valign="top" align="center">258</td>
<td valign="top" align="center">43,421</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s3_2">
<title>3.2 Proposed Method</title>
<p>The architecture of our SLP-Net is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>, which consists of two main components: sequence CNNs, and ConvLSTM. The sequence CNNs with multi-scale receptive fields are employed to extract more discriminative high-level features from the CT sequence, while the ConvLSTM captures the axial dimensional dynamics of features. In addition, a sequence attention map is utilized as an auxiliary means to integrate the output of the network and obtain prediction results with a higher level of confidence. Finally, an adaptive-weighted cross-entropy (ACE) loss is used to optimize the whole model.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Flowchart of our whole system for differentiating between viral pneumonia and bacterial pneumonia in a chest CT volume. Each volume is divided into overlapping sequences containing <italic>n</italic> slices during the training phase, such that the overlapping slices between two sequences are <italic>k</italic>. When predicting each volume during the testing phase, in addition to using the model to obtain the classification results of the sequence, we also introduce sequence attention maps for auxiliary classification to enhance the confidence level of the results. GT, ground truth; ACE, adaptive-weighted cross-entropy loss; FC, fully connected layer.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-11-781798-g002.tif"/>
</fig>
<sec id="s3_2_1">
<title>3.2.1 Sequence CNN With Multi-Scale Receptive Fields</title>
<p>A typical CNN model consists of a stack of convolution layers, interleaved with non-linear downsampling operations (e.g., max pooling) and point-wise nonlinearities (e.g., ReLU). The residual shortcut used in ResNet can reduce the over-fitting of the model, so that the depth of the network can be greater and achieve better performance. Taking into consideration the problems of overfitting and parameter cost, we employed ResNet (<xref ref-type="bibr" rid="B36">36</xref>) as our encoder backbone. The first four feature-extracting blocks are retained, without the average-pooling layer and the fully-connected layers.</p>
<p>Since VP and BP reveal similar appearances in CT images, we aim to obtain more discriminative features by employing multi-scale information, in order to distinguish them more accurately. Unlike most existing methods (<xref ref-type="bibr" rid="B37">37</xref>&#x2013;<xref ref-type="bibr" rid="B39">39</xref>) that improve multi-scale ability by utilizing features with different resolutions, we apply a recently proposed multi-scale receptive fields technique (<xref ref-type="bibr" rid="B40">40</xref>) to enhance representation ability at a more granular level. Specifically, we apply a modified bottleneck with multi-scale ability, the Res2Net module, to replace a group of 3 &#xd7; 3 filters used in the original bottleneck block of ResNet. As shown in <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref>, after the 1 &#xd7; 1 convolution, feature maps are split into <italic>s</italic> feature map subsets, denoted by <italic>x<sub>i</sub>
</italic>. Then, apart from <italic>x</italic>
<sub>1</sub>, each <italic>x<sub>i</sub>
</italic> goes through a corresponding 3 &#xd7; 3 convolutional operator, denoted by <italic>K<sub>i</sub>
</italic>(&#xb7;), where <italic>y<sub>i</sub>
</italic> is the output of <italic>K<sub>i</sub>
</italic>(&#xb7;). The output of <italic>K<sub>i&#x2013;1</sub>
</italic>(&#xb7;)is added to <italic>x<sub>i</sub>
</italic>, then sent to the next group of filters <italic>K<sub>i</sub>
</italic>(&#xb7;). Thus, <italic>y<sub>i</sub>
</italic> can be defined as follows:</p>
<disp-formula>
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo> <mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>;</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>K</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mo>&lt;</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>&#x2264;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow> </mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>A Res2Net module is utilized to extract more discriminative features.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-11-781798-g003.tif"/>
</fig>
<p>In order to better integrate the information from different scales, all outputs <italic>y<sub>i</sub>
</italic>, where <italic>i</italic> &#x2208;{1,2, &#x2026;, <italic>s</italic>}, are concatenated and passed through a 1&#xd7;1 convolution. Such splitting and concatenation strategies can force the convolution to process features more efficiently. Note that each 3&#xd7;3 convolution operation <italic>K<sub>i</sub>
</italic>(&#xb7;)receives information from all the feature splits {<italic>x<sub>j</sub>
</italic>, <italic>j</italic> &#x2264; <italic>i</italic>}. Each time <italic>x<sub>j</sub>
</italic> performs a 3&#xd7;3 convolution, the size of the receptive field will increase. Due to the combinatorial effect, the output of the Res2Net block contains different combinations of receptive field sizes/scales.</p>
</sec>
<sec id="s3_2_2">
<title>3.2.2 ConvLSTM With ACE Loss</title>
<p>Although the conventional fully-connected LSTM (FC-LSTM) can handle sequences of any length and capture long-term dependencies (<xref ref-type="bibr" rid="B41">41</xref>), it contains too much redundancy for spatial data, which is a critical problem for image sequences. Inspired by video object detection (<xref ref-type="bibr" rid="B42">42</xref>), we apply ConvLSTM (<xref ref-type="bibr" rid="B43">43</xref>) to process the feature sequences from the encoder.</p>
<p>As the convolutional counterpart of the FC-LSTM, the ConvLSTM introduces the convolution operation into the input-to-state and state-to-state transitions. The ConvLSTM can model axial dimensional dependencies while preserving spatial information. As with the FC-LSTM, the ConvLSTM unit (see <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>) includes an input gate <italic>i<sub>t</sub>
</italic>, a memory cell <italic>C<sub>t</sub>
</italic>, a forget gate <italic>f<sub>t</sub>
</italic> and an output gate <italic>o<sub>t</sub>
</italic>. The memory cell <italic>C<sub>t</sub>
</italic>, acting as an accumulator of the state information, is accessed, updated and cleared through self-parameterized controlling gates: <italic>i<sub>t</sub>
</italic>, <italic>o<sub>t</sub>
</italic>, and <italic>f<sub>t</sub>
</italic>. If the input gate is switched on, the new data is accumulated into the memory cell once an input arrives. Similarly, the past cell status <italic>C<sub>t&#x2013;</sub>
</italic>
<sub>1</sub> will be forgotten if the forget gate <italic>f<sub>t</sub>
</italic> is activated. The output gate <italic>o<sub>t</sub>
</italic> further controls whether the latest memory cell&#x2019;s value <italic>C<sub>t</sub>
</italic> will be transmitted to the final state <italic>H<sub>t</sub>
</italic>. With the above definitions, the ConvLSTM can be formulated as follows:</p>
<disp-formula>
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>f</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>i</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>tanh</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>c</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>W</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>&#x210b;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>tanh</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>ConvLSTM is utilized to implicitly learn axial dimensional dynamics and efficiently fuse axial dimensional features.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-11-781798-g004.tif"/>
</fig>
<p>where &#x2018;*&#x2019; denotes the convolution operator, &#x2018;&#xb0;&#x2019; denotes the Hadamard product, and <italic>&#x3c3;</italic> is the sigmoid activation function. <italic>X<sub>t</sub>
</italic> and <italic>&#x210b;<sub>t</sub>
</italic> are the input and output of the ConvLSTM at time step <italic>t</italic> (<italic>t</italic> indicates the <italic>t</italic>th frame in a CT image sequence, and slices will be referred to as frames in the sequel.), and <italic>i<sub>t</sub>
</italic>, <italic>f<sub>t</sub>
</italic>, and <italic>o<sub>t</sub>
</italic> indicate the input, forget and output gates, respectively. <italic>b<sub>i</sub>
</italic>, <italic>b<sub>f</sub>
</italic>, and <italic>b<sub>o</sub>
</italic> are the bias of the input gate, forget gate, and output gate. A memory cell <italic>C<sub>t</sub>
</italic> stores the historical information. All the gates <italic>i</italic>, <italic>f</italic>, <italic>o</italic>, memory cell <italic>C</italic>, hidden state &#x210b; and the learnable weights <italic>W</italic> are 3D tensors. Input sequences &#x1d4b3; are fed into a ConvLSTM block, which captures the long and short-term memory of sequences and contains both axial dimensional information, for use in implicitly learning axial dimensional dynamics and efficiently fusing axial dimensional features.</p>
<p>We define <italic>L<sub>t</sub>
</italic> as the output of the ConvLSTM layer at time step <italic>t</italic>. The output of the ConvLSTM layer is fed to the fully-connected (FC) layers, which transform the features into a space that makes the output easier to classify. The outputs of the FC layers are defined as <italic>O<sub>t</sub>
</italic> at time step <italic>t</italic>. Ideally, the longer the image sequence, and the more classification information ConvLSTM processes, the higher the confidence of classification. From this perspective, it is sufficient to use the output of the final time step for classification without further processing. However, in practice, due to differences in the distribution of lesions on different slices, there may be some useful information that has not been accumulated in the memory cell. In order to enchance the memory ablity of ConvLSTM for CT sequence at different slices and ensure that as much valid information from the previous slices as possible are encoded, we propose to use all the intermediate outputs of every time step as our feature for identification. A better ConvLSTM means that the longer the sequence it processes and the more comprehensive information it considers, the more confident it identifies the input. From this perspective, instead of minimizing the loss on the final time step, we define a new adaptive-weighted cross-entropy (ACE) loss to use all the intermediate outputs of every time step weighted by <italic>w<sub>t</sub>
</italic>:</p>
<disp-formula>
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>&#x2112;</mml:mi>
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>E</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo stretchy="false">[</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mi>log</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>O</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>C</italic> and <italic>p</italic> denote the classifier and classification label, respectively, and <italic>n</italic> denotes the number of images in a sequence. <italic>C<sub>p</sub>
</italic>(<italic>O<sub>t</sub>
</italic>) indicates the classifier <italic>C</italic>, which correctly identifies the final output <italic>O</italic> at time step <italic>t</italic>, <italic>y<sub>p</sub>
</italic> &#x2208;{0, 1, 2} are the label values; and <italic>P</italic>=3 denotes the total number of labels. Finally, <italic>w<sub>t</sub>
</italic> is the weight of each frame in a sequence. weight. We let each group of two weight items constitute the arithmetic sequence.</p>
<p>Since the importance of the information contained in different slices is different, it is not reasonable to use the equal weights. Due to the output of the final time step has taken into account all other previous slices , it contains the most information, and the further away from the last output, the less information it contains. Moreover, the number of slices in a sequence is a hyper-parameter, we adopt an adaptive weighting scheme. The output of the final time step should be assigned the maximum weight, and the farther away from the final time step, the smaller the weight. Specifically, we let each group of two weight items constitute the arithmetic sequence: the sum of which is 1. The first two items are taken as 0.01, namely, <italic>w</italic>
<sub>1</sub> = <italic>w</italic>
<sub>2</sub> = 0.01, and the subsequent weights can then be calculated according to the hyperparameter <italic>n</italic> and weight <italic>w</italic>
<sub>1</sub>. The ACE loss ensures that the features of the previous CT images in the sequence can be encoded into the later image.</p>
</sec>
<sec id="s3_2_3">
<title>3.2.3 Auxiliary Diagnosis With Attention Maps</title>
<p>Deciding which type (VP, BP, or normal) the entire volume belongs to based on the prediction results of the sequence is a critical step in auxiliary diagnosis. For higher confidence, in addition to using the model to obtain the classification result of the sequence, we also utilized the Grad-CAM (<xref ref-type="bibr" rid="B44">44</xref>) technology to generate attention maps of the sequence to assist the prediction. Grad-CAM is a method for producing visual interpretations for CNNs in the form of class-specific saliency maps. A saliency map, <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msubsup>
<mml:mi>L</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>c</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, is produced for each image input based on the activation from <italic>k</italic> filters, <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:math>
</inline-formula>, at the final convolutional layer. To make the method applicable to image sequences, the activations for all timesteps <italic>t</italic> in the sequence are considered (Eq. R1).</p>
<disp-formula>
<label>(R1)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msubsup>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mi>a</mml:mi>
</mml:munder>
<mml:mrow>
<mml:msubsup>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mstyle>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
<mml:mo>;</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>Z</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munder>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msup>
<mml:mi>F</mml:mi>
<mml:mi>c</mml:mi>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2202;</mml:mo>
<mml:msubsup>
<mml:mi>A</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
<mml:mi>k</mml:mi>
</mml:msubsup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>Z</italic> is a normalizing constant and <italic>F<sup>c</sup>
</italic> is the network output for the class <italic>c</italic>. <italic>i, j</italic> are pixel location of filter <italic>A<sup>k</sup>
</italic>. In the visualization examples shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, stronger class activation map (CAM) areas are indicated with lighter colors.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Examples of attention maps obtained with Grad-CAM. <bold>(A)</bold> Viral pneumonia cases. <bold>(B)</bold> Bacterial pneumonia cases. <bold>(C)</bold> Normal cases. Lighter colors indicate the stronger response regions. From the maps, the infected regions receive greater attention.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-11-781798-g005.tif"/>
</fig>
<p>During the prediction phase, we can obtain the classification results of sequences belonging to a volume. In addition, we apply Grad-CAM to generate the response heat map of each sequence. A volume containing <italic>m</italic> sequences will be classified as viral pneumonia (VP) sample if it meets both of following criteria: (a) More sequences are classified as viral pneumonia (VP) then bacterial pneumonia (BP) in this volume; (b) There are two adjacent sequences with the category of viral pneumonia whose activation regions have an intersecting area of more than 50%. If the second criterion is not satisfied for VP, the bacterial type sequences are checked if there are two adjacent sequences with an intersecting area of more than 50% of the activation region, and if so the sample is classified as bacterial type, otherwise it is classified as normal. Notably, if there are the same number of VP and BP sequences, the one with the greater average sequence probability value is treated as the dominant category. The possibility output of the network dominates the classification on the volume level, and Attention Map is used as an auxiliary during the decision making.</p>
</sec>
</sec>
<sec id="s3_3">
<title>3.3 Evaluation Metrics</title>
<p>We employ the commonly used metrics for multi-class classification to measure performance: e.g., weighted sensitivity (Sen, also known as recall), specificity (Spe), accuracy (Acc), and balanced accuracy (B-Acc, a.k.a. balanced classification rate). In order to reflect the tradeoff between sensitivity and specificity, and evaluate the quality of our classification results more reliably, a kappa analysis and F-measure (<italic>F</italic>1 score) are also provided following (<xref ref-type="bibr" rid="B45">45</xref>). These two measures are more robust than other percentage agreement measures, as they take into account the possibility of the agreement occurring by chance. The weighted sensitivity (Sen), specificity (Spe), accuracy (Acc), and balanced accuracy (B-Acc) are defined as:</p>
<disp-formula>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mstyle>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>S</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:mstyle>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>A</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>c</mml:mi>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mstyle>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>N</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where TP<italic>
<sub>i</sub>
</italic> indicates the number of true positives, TN<italic>
<sub>i</sub>
</italic>&#x2014;the number of true negatives, FP<italic>
<sub>i</sub>
</italic>&#x2014;the number of false positives, and FN<italic>
<sub>i</sub>
</italic>&#x2014;the number of false negatives for the i&#x2013;<italic>th</italic> classification label; and w<italic>
<sub>i</sub>
</italic> represents the percentage of images whose ground truth labels are <italic>i</italic>. The kappa values and F-measure (<italic>F</italic>1 score, a.k.a. Dice score) are defined as follows:</p>
<disp-formula>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mi>n</mml:mi>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mstyle>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2217;</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>n</mml:mi>
<mml:mo>&#x2217;</mml:mo>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>=</mml:mo>
</mml:mrow>
</mml:mstyle>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>P</mml:mi>
</mml:munderover>
<mml:mrow>
<mml:msub>
<mml:mi>w</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>a</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where a<italic>
<sub>i</sub>
</italic> denotes the true sample number of each class, b<italic>
<sub>i</sub>
</italic> denotes the predicted sample number of each class, <italic>n</italic> denotes the total sample number, and <italic>P</italic> denotes the number of classes. Note that <italic>kappa</italic> values between 0.81 to 1.00 indicate almost perfect agreement, values between 0.61 and 0.80 exhibit substantial agreement, values of 0.41&#x2013;0.60 exhibit moderate agreement and values less than 0.40 exhibit poor to fair agreement. The <italic>F</italic>1 score reaches its best value at 1 and worst at 0. We also present the ROC curves and the area under ROC curve (AUC) for VP against BP.</p>
</sec>
<sec id="s3_4">
<title>3.4 Implementation Details</title>
<p>The proposed method was implemented in the publicly available Pytorch library. The combination of CNN and ConvLSTM makes the model more complex. To accelerate convergence, we first trained a CNN classification network with a labeled 2D slice. After removing the FC layer, the encoder is used as the initialization parameter of SLP-Net. During the training phase of SLP-Net, CNN and ConvLSTM are jointly trained in an end-to-end manner using Adam optimizer. In practice, we found that CNN pre-training does speed up the convergence of the model, but has no effect on the final classification performance. The learning rate was gradually decreasing starting from 0.0001, and the momentum was set to 0.9. In addition, online data enhancement was employed to enlarge the training sequence data. The same data enhancement was used for all images in a sequence: we implemented data augmentation in a random way, including brightness, color, contrast, and sharpness transformation from 90 to 110%. We set a random seed from 1 to 4 for the enhancement.</p>
</sec>
</sec>
<sec id="s4">
<title>4 Results</title>
<sec id="s4_1">
<title>4.1 Classification Performances</title>
<p>To compare the classification performance, we evaluated the detection ability of the model at both sequence and volume levels. All the existing pneumonia detection methods are accomplished using 2D CT slices or 3D volumes. To further verify whether the features containing both axial dimensional and spatial information captured by our model could benefit detection performance, we compared the proposed method to other classic classification models using 2D slices: AlexNet (<xref ref-type="bibr" rid="B46">46</xref>), VGG19 (<xref ref-type="bibr" rid="B47">47</xref>), InceptionV3 (<xref ref-type="bibr" rid="B48">48</xref>), ResNet34 (<xref ref-type="bibr" rid="B36">36</xref>), and Xception (<xref ref-type="bibr" rid="B49">49</xref>). Due to the lack of sufficient training data and the GPU memory constraint, we cannot apply 3D CNNs on complete CT volumes. In order to compare the proposed SLP-Net with other 3D deep learning architectures, we apply CT sequence data, which can be considered as 3D data, to train 3D models, including C3D (<xref ref-type="bibr" rid="B50">50</xref>), I3D (<xref ref-type="bibr" rid="B51">51</xref>), and S3D (<xref ref-type="bibr" rid="B52">52</xref>). We report the detection results for slice/sequence-level and case-level in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>. We applied a similar strategy to that in <italic>Auxiliary Diagnosis With Attention Maps</italic> section to determine the prediction result of a volume when using the 2D models. First, Grad-CAM was used to generate the activated maps of 2D slices, and binary activated maps can be obtained through thresholding. If five consecutive slices in a volume were predicted as indicative of viral pneumonia, and the intersection area of their activated area exceeds 50% of the union area, the volume was considered to be indicative of viral pneumonia.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Classification results for VP, BP and normal controls by different methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Method </th>
<th valign="top" colspan="5" align="center">Slice/Sequence-Level</th>
<th valign="top" colspan="5" align="center">Case-level</th>
</tr>
<tr>
<th valign="top" align="left"> </th>
<th valign="top" align="center">Kappa</th>
<th valign="top" align="center">F1</th>
<th valign="top" align="center">B-Acc</th>
<th valign="top" align="center">Sen</th>
<th valign="top" align="center">Spe</th>
<th valign="top" align="center">Kappa</th>
<th valign="top" align="center">F1</th>
<th valign="top" align="center">B-Acc</th>
<th valign="top" align="center">Sen</th>
<th valign="top" align="center">Spe</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AlexNet </td>
<td valign="top" align="center">0.5207</td>
<td valign="top" align="center">0.5680</td>
<td valign="top" align="center">0.6872</td>
<td valign="top" align="center">0.6375</td>
<td valign="top" align="center">0.7370</td>
<td valign="top" align="center">0.6889</td>
<td valign="top" align="center">0.7381</td>
<td valign="top" align="center">0.8207</td>
<td valign="top" align="center">0.8274</td>
<td valign="top" align="center">0.8140</td>
</tr>
<tr>
<td valign="top" align="left">VGG19 </td>
<td valign="top" align="center">0.6258</td>
<td valign="top" align="center">0.6574</td>
<td valign="top" align="center">0.7502</td>
<td valign="top" align="center">0.7152</td>
<td valign="top" align="center">0.7853</td>
<td valign="top" align="center">0.7709</td>
<td valign="top" align="center">0.8000</td>
<td valign="top" align="center">0.8571</td>
<td valign="top" align="center">0.8690</td>
<td valign="top" align="center">0.8601</td>
</tr>
<tr>
<td valign="top" align="left">ResNet34 </td>
<td valign="top" align="center">0.6767</td>
<td valign="top" align="center">0.7100</td>
<td valign="top" align="center">0.7948</td>
<td valign="top" align="center">0.7783</td>
<td valign="top" align="center">0.8112</td>
<td valign="top" align="center">0.8489</td>
<td valign="top" align="center">0.8598</td>
<td valign="top" align="center">0.9045</td>
<td valign="top" align="center">0.9048</td>
<td valign="top" align="center">0.9043</td>
</tr>
<tr>
<td valign="top" align="left">InceptionV3 </td>
<td valign="top" align="center">0.5177</td>
<td valign="top" align="center">0.6107</td>
<td valign="top" align="center">0.7156</td>
<td valign="top" align="center">0.6978</td>
<td valign="top" align="center">0.7333</td>
<td valign="top" align="center">0.7692</td>
<td valign="top" align="center">0.7976</td>
<td valign="top" align="center">0.8607</td>
<td valign="top" align="center">0.8631</td>
<td valign="top" align="center">0.8582</td>
</tr>
<tr>
<td valign="top" align="left">Xception </td>
<td valign="top" align="center">0.6802</td>
<td valign="top" align="center">0.6776</td>
<td valign="top" align="center">0.7688</td>
<td valign="top" align="center">0.7252</td>
<td valign="top" align="center">0.8124</td>
<td valign="top" align="center">0.8500</td>
<td valign="top" align="center">0.8631</td>
<td valign="top" align="center">0.9048</td>
<td valign="top" align="center">0.9107</td>
<td valign="top" align="center">0.9061</td>
</tr>
<tr>
<td valign="top" align="left">C3D </td>
<td valign="top" align="center">0.7382</td>
<td valign="top" align="center">0.7442</td>
<td valign="top" align="center">0.8232</td>
<td valign="top" align="center">0.8013</td>
<td valign="top" align="center">0.8450</td>
<td valign="top" align="center">0.8616</td>
<td valign="top" align="center">0.8729</td>
<td valign="top" align="center">0.9158</td>
<td valign="top" align="center">0.9183</td>
<td valign="top" align="center">0.9132</td>
</tr>
<tr>
<td valign="top" align="left">I3D </td>
<td valign="top" align="center">0.7437</td>
<td valign="top" align="center">0.7513</td>
<td valign="top" align="center">0.8403</td>
<td valign="top" align="center">0.8302</td>
<td valign="top" align="center">0.8132</td>
<td valign="top" align="center">0.8481</td>
<td valign="top" align="center">0.8952</td>
<td valign="top" align="center">0.9229</td>
<td valign="top" align="center">0.9167</td>
<td valign="top" align="center">0.9290</td>
</tr>
<tr>
<td valign="top" align="left">S3D </td>
<td valign="top" align="center">0.7525</td>
<td valign="top" align="center">0.7332</td>
<td valign="top" align="center">0.8106</td>
<td valign="top" align="center">0.7696</td>
<td valign="top" align="center">0.8517</td>
<td valign="top" align="center">0.8696</td>
<td valign="top" align="center">0.8796</td>
<td valign="top" align="center">0.9297</td>
<td valign="top" align="center">0.9133</td>
<td valign="top" align="center">0.9157</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>SLP-Net</bold>
</td>
<td valign="top" align="center">
<bold>0.8280</bold>
</td>
<td valign="top" align="center">
<bold>0.8123</bold>
</td>
<td valign="top" align="center">
<bold>0.8665</bold>
</td>
<td valign="top" align="center">
<bold>0.8397</bold>
</td>
<td valign="top" align="center">
<bold>0.8934</bold>
</td>
<td valign="top" align="center">
<bold>0.9263</bold>
</td>
<td valign="top" align="center">
<bold>0.9291</bold>
</td>
<td valign="top" align="center">
<bold>0.9523</bold>
</td>
<td valign="top" align="center">
<bold>0.9524</bold>
</td>
<td valign="top" align="center">
<bold>0.9521</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The best performance of all the methods is highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>As can be observed, the 3D networks achieve higher performances than the 2D networks, which confirms the importance of the combination of axial dimensional and spatial information for accurate detection results. At a slice/sequence level, our SLP-Net outperformed other methods in terms of <italic>kappa</italic>, <italic>F</italic>1 and <italic>Sen</italic> by a large margin, as well as achieving the best performance at a volume level. In addition, <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> shows the confusion matrices of VGG-19, Resnet34, Xception, C3D, S3D, and our method over the dataset. These results further indicate the superiority of the performance of our approach. As stated in the Introduction section, the difficulty of pneumonia diagnosis is the differentiation between BP and VP. Accordingly, we conducted experiments on the dataset contained VP and BP samples only. Results are shown in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> and <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. We may observe that the proposed method again produces the best performance compared to the other methods. <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref> gathers all the performances of these models. The results show that the 3D-based method is generally better than the 2D-based method, mainly because the 3D input provides richer spatial information, which allows the model to learn and extract the subtle differences in the spatial distribution of different diseases, which is especially important for difficult samples with similar lesion appearance. In this regard, our proposed method not only utilizes the 3D information, but also explicitly focuses on the lesion area in the decision-making process through attention map, which makes the classification results more reliable. This idea can be applied to many different medical image-based classification tasks, since the similarity of lesion appearance is a problem in many scenarios.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Confusion matrices of the different methods at slice/sequence level. <bold>(A&#x2013;F)</bold> are the results of VGG19, Resnet34, Xception, C3D, S3D and ours, respectively. The numbers in the confusion matrices denote the percentage (above) and number (below) of the predicted class.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-11-781798-g006.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Comparison of different methods in classifying viral pneumonia (VP) and bacterial pneumonia (BP), at a slice/sequence level.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Method</th>
<th valign="top" align="center">Acc</th>
<th valign="top" align="center">Sen</th>
<th valign="top" align="center">Spe</th>
<th valign="top" align="center">AUC (<italic>p</italic>-value)</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">AlexNet</td>
<td valign="top" align="center">0.7327</td>
<td valign="top" align="center">0.8182</td>
<td valign="top" align="center">0.6650</td>
<td valign="top" align="center">0.8700 (<italic>p</italic> &lt;0.001)</td>
</tr>
<tr>
<td valign="top" align="left">VGG19</td>
<td valign="top" align="center">0.7776</td>
<td valign="top" align="center">0.8197</td>
<td valign="top" align="center">0.7442</td>
<td valign="top" align="center">0.8785 (<italic>p</italic> &lt;0.001)</td>
</tr>
<tr>
<td valign="top" align="left">ResNet34</td>
<td valign="top" align="center">0.7939</td>
<td valign="top" align="center">0.8305</td>
<td valign="top" align="center">0.7649</td>
<td valign="top" align="center">0.8874 (<italic>p</italic> &lt;0.001)</td>
</tr>
<tr>
<td valign="top" align="left">InceptionV3</td>
<td valign="top" align="center">0.7429</td>
<td valign="top" align="center">0.8028</td>
<td valign="top" align="center">0.6955</td>
<td valign="top" align="center">0.8697 (<italic>p</italic> &lt;0.001)</td>
</tr>
<tr>
<td valign="top" align="left">Xception</td>
<td valign="top" align="center">0.8170</td>
<td valign="top" align="center">0.7704</td>
<td valign="top" align="center">0.8438</td>
<td valign="top" align="center">0.8874 (<italic>p</italic> &lt;0.001)</td>
</tr>
<tr>
<td valign="top" align="left">C3D</td>
<td valign="top" align="center">0.8218</td>
<td valign="top" align="center">
<bold>0.869</bold>
</td>
<td valign="top" align="center">0.7844</td>
<td valign="top" align="center">0.9129 (<italic>p</italic> &lt;0.05)</td>
</tr>
<tr>
<td valign="top" align="left">I3D</td>
<td valign="top" align="center">0.8320</td>
<td valign="top" align="center">0.8274</td>
<td valign="top" align="center">0.8356</td>
<td valign="top" align="center">0.8956 (<italic>p</italic> &lt;0.001)</td>
</tr>
<tr>
<td valign="top" align="left">S3D</td>
<td valign="top" align="center">0.8361</td>
<td valign="top" align="center">0.8413</td>
<td valign="top" align="center">0.8319</td>
<td valign="top" align="center">0.9035 (<italic>p</italic> &lt;0.01)</td>
</tr>
<tr>
<td valign="top" align="left">
<bold>SLP-Net</bold>
</td>
<td valign="top" align="center">
<bold>0.8476</bold>
</td>
<td valign="top" align="center">0.8459</td>
<td valign="top" align="center">
<bold>0.8490</bold>
</td>
<td valign="top" align="center">
<bold>0.9170</bold>
</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>P-value is calculated by Delong&#x2019;s test.</p>
<p>The best performance of all the methods is highlighted in bold.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>ROC curves in classifying viral pneumonia (VP) and bacterial pneumonia (BP) of the compared models and the proposed method.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-11-781798-g007.tif"/>
</fig>
<p>
<xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref> summarizes space and time cost of different methods. For fair comparison of inference time, we test all these models with PyTorch. Our SLP-Net had the best time efficiency and achieved smallest model size because it didn&#x2019;t use 3D convolution operations.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Model size and inference time of different methods.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left"/>
<th valign="top" align="center">C3D</th>
<th valign="top" align="center">I3D</th>
<th valign="top" align="center">S3D</th>
<th valign="top" align="center">SLP-Net</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Model size (MB)</td>
<td valign="top" align="center">39.2</td>
<td valign="top" align="center">48.7</td>
<td valign="top" align="center">42.3</td>
<td valign="top" align="center">34.4</td>
</tr>
<tr>
<td valign="top" align="left">Time (ms)</td>
<td valign="top" align="center">41.4</td>
<td valign="top" align="center">59.0</td>
<td valign="top" align="center">47.1</td>
<td valign="top" align="center">39.5</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s5">
<title>5 Analysis and Discussion</title>
<sec id="s5_1">
<title>5.1 Sensitivities to Hyperparameters</title>
<p>In <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> we investigate the sequence settings, i.e., <italic>n</italic> and <italic>k</italic>, denote the number of slices per sequence and the number of overlapping slices between two sequences, respectively. By default we set <italic>n</italic> = 10 and <italic>k</italic> = 5. As can be observed, the performance was greatly affected by the value assigned to <italic>n</italic>. When <italic>n</italic> is set very small (<italic>n</italic> = 5), the <italic>kappa</italic> and <italic>F</italic>1 drop by the considerable margin of 3%, demonstrating that the more axial dimensional information ConvLSTM encodes, the more the model benefits. However, when <italic>n</italic> is greater than 15, the model performance will decline. This may be due to the fact that as the slice number increases, there will be less training data, resulting in the model not being fully trained. <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref> also shows that our result is impacted just marginally when <italic>k</italic> is within a scale of 5-7. The performance of the model mainly depends on the abundance of the information contained in the sequence, that is, the more information contained in the sequence, the better the classification performance. Compared to n = 10, the sequence provide less timing information when n = 5, so the classification performance will decrease. If n is too large (e.g., n = 20), the performance will decrease due to fewer training samples.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Effect of different settings of hyperparameter <italic>n</italic> and <italic>k</italic> on the results.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Method</th>
<th valign="top" align="center">kappa</th>
<th valign="top" align="center">F1</th>
<th valign="top" align="center">Acc</th>
<th valign="top" align="center">B-Acc</th>
<th valign="top" align="center">Sen</th>
<th valign="top" align="center">Spe</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">n = 5, k = 3</td>
<td valign="top" align="center">0.7376</td>
<td valign="top" align="center">0.7469</td>
<td valign="top" align="center">0.8546</td>
<td valign="top" align="center">0.8272</td>
<td valign="top" align="center">0.8096</td>
<td valign="top" align="center">0.8449</td>
</tr>
<tr>
<td valign="top" align="left">n = 10, k = 3</td>
<td valign="top" align="center">0.7652</td>
<td valign="top" align="center">0.7701</td>
<td valign="top" align="center">0.8684</td>
<td valign="top" align="center">0.8453</td>
<td valign="top" align="center">0.8307</td>
<td valign="top" align="center">0.8599</td>
</tr>
<tr>
<td valign="top" align="left">n = 10, k = 5</td>
<td valign="top" align="center">0.8241</td>
<td valign="top" align="center">0.8091</td>
<td valign="top" align="center">0.9012</td>
<td valign="top" align="center">0.8641</td>
<td valign="top" align="center">0.8471</td>
<td valign="top" align="center">0.8911</td>
</tr>
<tr>
<td valign="top" align="left">n = 10, k = 7</td>
<td valign="top" align="center">0.8280</td>
<td valign="top" align="center">0.8123</td>
<td valign="top" align="center">0.9034</td>
<td valign="top" align="center">0.8665</td>
<td valign="top" align="center">0.8397</td>
<td valign="top" align="center">0.8934</td>
</tr>
<tr>
<td valign="top" align="left">n = 15, k = 5</td>
<td valign="top" align="center">0.7453</td>
<td valign="top" align="center">0.7557</td>
<td valign="top" align="center">0.8577</td>
<td valign="top" align="center">0.8362</td>
<td valign="top" align="center">0.8231</td>
<td valign="top" align="center">0.8493</td>
</tr>
<tr>
<td valign="top" align="left">n = 20, k = 5</td>
<td valign="top" align="center">0.7329</td>
<td valign="top" align="center">0.7068</td>
<td valign="top" align="center">0.8577</td>
<td valign="top" align="center">0.7930</td>
<td valign="top" align="center">0.7450</td>
<td valign="top" align="center">0.8410</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Here, <italic>n</italic> and <italic>k</italic> denote the number of slices in the sequence and the number of overlapping slices between two sequences, respectively.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
<sec id="s5_2">
<title>5.2 Ablation Study</title>
<p>Our SLP-Net employs three main components to form the classification framework: a sequence CNNs with multi-scale receptive fields, a ConvLSTM module, and a carefully designed ACE loss. In this subsection, we analyze and discuss the network under different scenarios to validate the performance of each key component of our model, and the results of different combinations of these modules are reported in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Ablation studies of our SLP-Net.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fonc-11-781798-g008.tif"/>
</fig>
<sec id="s5_2_1">
<title>5.2.1 Effectiveness of ConvLSTM</title>
<p>To explore the contribution of the ConvLSTM, we use a ResNet50 pretrained on ImageNet as the backbone. As shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, a backbone + ConvLSTM + Res2Net method clearly outperformes the backbone + Res2Net, with improvement of about 6% in <italic>F</italic>1. This shows that the ConvLSTM is capable of extracting the axial dimensional and spatial information, thus memorizing the change in appearance that corresponds to axial dimensional information, and improving the performance in identifying and discriminating between VP and BP.</p>
</sec>
<sec id="s5_2_2">
<title>5.2.2 Effectiveness of the Res2Net module</title>
<p>We investigated the importance of the multi-scale sequence module, i.e., Res2Net. From <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, we observe that a backbone + ConvLSTM + Res2Net model outperformed the backbone model in terms of major metrics, i.e., <italic>kappa</italic> and <italic>F</italic>1. This suggests that introducing the Res2Net module enables the encoder to capture more discriminative features to accurately differentiate VP from BP.</p>
</sec>
<sec id="s5_2_3">
<title>5.2.3 Effectiveness of ACE</title>
<p>Finally, we investigate the importance of the ACE loss. From the results in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>, it may clearly be observed that the ACE Loss effectively improves the classification performance in our model. One possible reason is that, with the ACE loss, the ConvLSTM explores the axial dimensional dynamics of appearance features in CT sequences, and these features are further aggregated for classification purposes.</p>
</sec>
<sec id="s5_2_4">
<title>5.2.4 Effectiveness of Attention Maps</title>
<p>To investigate the contribution of the Attention Map, we added an additional experiment&#x2014;case-level classification without Attention Map. Specifically, sequence-level classification results were first obtained using SLP-Net, and if there were VP or BP sequences in a volume, the type with more number is used as the category of the whole volume. If it does not contain VP and BP sequence, it is classified as normal. Notably, if there are the same number of VP and BP sequences, the one with the greater average sequence probability value is treated as the dominant category. <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref> shows the result, where SLP-Net with Attention Map as auxiliary achieves better performance than without Attention Map. This demonstrates that with the aid of attention map, the distribution of lesions can be considered simultaneously in the decision-making process, thus improving the performance of case-level classification.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Ablation study of Attention Map in classifying VP, BP, and Normal controls at the case-level.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Method </th>
<th valign="top" align="center">Kappa</th>
<th valign="top" align="center">F1</th>
<th valign="top" align="center">B-Acc</th>
<th valign="top" align="center">Sen</th>
<th valign="top" align="center">Spe</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Without Attention Map</td>
<td valign="top" align="center">0.8731</td>
<td valign="top" align="center">0.9090</td>
<td valign="top" align="center">0.9167</td>
<td valign="top" align="center">0.9174</td>
<td valign="top" align="center">0.9160</td>
</tr>
<tr>
<td valign="top" align="left">With Attention Map</td>
<td valign="top" align="center">0.9263</td>
<td valign="top" align="center">0.9291</td>
<td valign="top" align="center">0.9523</td>
<td valign="top" align="center">0.9524</td>
<td valign="top" align="center">0.9521</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
</sec>
<sec id="s6">
<title>6 Discussion and Conclusions</title>
<sec id="s6_1">
<title>6.1 Limitations</title>
<p>Although our method achieves better results in the pneumonia classification task compared to other methods, this work still has some limitations. Firstly, we used the multiscale feature technique Res2Net in the feature extraction part, but did not further explore the hyperparameter settings in it, and although we believe that careful selection of hyperparameters may further improve the classification performance, no additional experiments were conducted in this work to compare the impact of different hyperparameters since this is not the focus of our work. Secondly, the model is not evaluated on an external dataset. To our knowledge, there are no publicly available 3D CT datasets for different types of pneumonia classification tasks, and it is difficult to collect compliant data from multiple centers due to various conditions. We intend to evaluated the performance of our model on external datasets in the future.</p>
</sec>
<sec id="s6_2">
<title>6.2 Conclusion</title>
<p>Hospitals are beginning to use CT imaging in the diagnosis of viral pneumonia, and it is vital to improve the sensitivity of diagnostic methods so as to reduce the incidence of false negatives. AI-empowered image acquisition workflows are effective, and may also aid in protecting clinicians from viral pneumonia (e.g., COVID-19) infection. Although several effective AI-based COVID-19 diagnosis or lesion segmentation methods have been introduced recently, automated differentiation of viral pneumonia from other types of pneumonia is still a challenging task. The motivation of this study was to employ AI techniques to alleviate the problem posed by the fact that even radiologists are hard pressed to distinguish VP from BP, as they share very similar presentations of infection lesion characteristics in CT images.</p>
<p>In this paper, we have proposed a novel viral pneumonia detection network, named SLP-Net. By contrast with previous 2D slice-based or 3D volume-based methods, we considered continuous CT images as time sequences. Our model first utilized the sequence CNNs with multi-scale receptive fields to extract a sequence of higher-level representations. The feature sequences were then fed into a ConvLSTM to capture axial dimensional features. Finally, in order to ensure that as many valid features from previous slice as possible are encoded into the later CT slices, a novel ACE loss was proposed to optimize the output of the SLP-Net. Furthermore, during the prediction phase, we used sequence attention maps for auxiliary classification to predict each volume, which can enhance the confidence level of the results. Overall, in order to accurately distinguish VP from BP and normal subjects, we used the sequence CNNs with multi-scale receptive fields to extract more differentiating features, and then applied a ConvLSTM to capture axial dimensional features of the CT sequence, thereby obtaining features containing both axial dimensional and spatial information. The superior evaluation performance achieved in the classification experiments demonstrate the ability of our model in the differential diagnosis of VP, BP and normal cases. Although we only evaluated our method on the CT dataset of pneumonia, it can be adapted to any other 3D medical image classification problems, such as lung cancer imaging analysis, and the identification of Alzheimer&#x2019;s disease. In future work we will further validate our models on even larger datasets, and seek its implementation in real clinical settings.</p>
</sec>
</sec>
<sec id="s7" sec-type="data-availability">
<title>Data Availability Statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding author.</p>
</sec>
<sec id="s8" sec-type="ethics-statement">
<title>Ethics Statement</title>
<p>The studies involving human participants were reviewed and approved by the Hwa Mei Hospital, University of Chinese Academy of Sciences. The patients/participants provided their written informed consent to participate in this study.</p>
</sec>
<sec id="s9" sec-type="author-contributions">
<title>Author Contributions</title>
<p>JH, JX, and RL were involved in data analysis and interpretation, and drafting and revising the manuscript. HH, YM, KY, RRL, YLZ, and JJZ were involved in data analysis and interpretation. JL, JFZ, and YTZ were involved in study conceptualization, supervision, revising the manuscript. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec id="s10" sec-type="funding-information">
<title>Funding</title>
<p>This work was supported in part by the Zhejiang Provincial Natural Science Foundation of China (LZ19F010001, and LQ20F030002), in part by the Key Project of Ningbo Public Welfare Science and Technology (2021S107), and in part by the Youth Innovation Promotion Association CAS (2021298).</p>
</sec>
<sec id="s11" sec-type="COI-statement">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ref-list>
<title>References</title>
<ref id="B1">
<label>1</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Horby</surname> <given-names>PW</given-names>
</name>
<name>
<surname>Hayden</surname> <given-names>FG</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>GF</given-names>
</name>
</person-group>. <article-title>A Novel Coronavirus Outbreak of Global Health Concern</article-title>. <source>Lancet</source> (<year>2020</year>) <volume>395</volume>:<page-range>470&#x2013;3</page-range>. doi: <pub-id pub-id-type="doi">10.1016/S0140-6736(20)30185-9</pub-id>
</citation>
</ref>
<ref id="B2">
<label>2</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>C</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>L</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>J</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Clinical Features of Patients Infected With 2019 Novel Coronavirus in Wuhan, China</article-title>. <source>Lancet</source> (<year>2020</year>) <volume>395</volume>:<fpage>497</fpage>&#x2013;<lpage>506</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0140-6736(20)30183-5</pub-id>
</citation>
</ref>
<ref id="B3">
<label>3</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ai</surname> <given-names>T</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Hou</surname> <given-names>H</given-names>
</name>
<name>
<surname>Zhan</surname> <given-names>C</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>C</given-names>
</name>
<name>
<surname>Lv</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>Correlation of Chest CT and RT-PCR Testing in Coronavirus Disease 2019 (COVID-19) in China: A Report of 1014 Cases</article-title>. <source>Radiology</source> (<year>2020</year>) <volume>296</volume>:<page-range>E32&#x2013;40</page-range>. doi: <pub-id pub-id-type="doi">10.1148/radiol.2020200642</pub-id>
</citation>
</ref>
<ref id="B4">
<label>4</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ying</surname> <given-names>L</given-names>
</name>
<name>
<surname>Pang</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Sensitivity of Chest CT for COVID-19: Comparison to RT-PCR</article-title>. <source>Radiology</source> (<year>2020</year>) <volume>296</volume>:<page-range>E115&#x2013;7</page-range>. doi: <pub-id pub-id-type="doi">10.1148/radiol.2020200432</pub-id>
</citation>
</ref>
<ref id="B5">
<label>5</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ying</surname> <given-names>L</given-names>
</name>
<name>
<surname>Pang</surname> <given-names>P</given-names>
</name>
<etal/>
</person-group>. <article-title>Sensitivity of Chest Ct for Covid-19: Comparison to Rt-Pcr</article-title>. <source>Radiology</source> (<year>2020</year>) <volume>296</volume>:<page-range>E115&#x2013;7</page-range>. doi: <pub-id pub-id-type="doi">10.1148/radiol.2020200432</pub-id>
</citation>
</ref>
<ref id="B6">
<label>6</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pan</surname> <given-names>F</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>T</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>P</given-names>
</name>
<name>
<surname>Gui</surname> <given-names>S</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>B</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L</given-names>
</name>
<etal/>
</person-group>. <article-title>Time Course of Lung Changes at Chest CT During Recovery From Coronavirus Disease 2019 (COVID-19)</article-title>. <source>Radiology</source> (<year>2020</year>) <volume>295</volume>:<page-range>715&#x2013;21</page-range>. doi: <pub-id pub-id-type="doi">10.1148/radiol.2020200370</pub-id>
</citation>
</ref>
<ref id="B7">
<label>7</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>McLuckie</surname> <given-names>A</given-names>
</name>
</person-group>. <source>Respiratory Disease and its Management</source>. <publisher-loc>London, UK</publisher-loc>: <publisher-name>Springer Science &amp; Business Media</publisher-name> (<year>2009</year>).</citation>
</ref>
<ref id="B8">
<label>8</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miller</surname> <given-names>WT</given-names>
</name>
<name>
<surname>Mickus</surname> <given-names>TJ</given-names>
</name>
<name>
<surname>Barbosa</surname> <given-names>JE</given-names>
</name>
<name>
<surname>Mullin</surname> <given-names>C</given-names>
</name>
<name>
<surname>Van</surname> <given-names>VM</given-names>
</name>
<name>
<surname>Shiley</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>CT of Viral Lower Respiratory Tract Infections in Adults: Comparison Among Viral Organisms and Between Viral and Bacterial Infections</article-title>. <source>Am J Roentgenology</source> (<year>2011</year>) <volume>197</volume>:<page-range>1088&#x2013;95</page-range>. doi: <pub-id pub-id-type="doi">10.2214/AJR.11.6501</pub-id>
</citation>
</ref>
<ref id="B9">
<label>9</label>
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Jae</surname> <given-names>YP</given-names>
</name>
<name>
<surname>Rosemary</surname> <given-names>F</given-names>
</name>
<name>
<surname>Richard</surname> <given-names>S</given-names>
</name>
<name>
<surname>Neil</surname> <given-names>S</given-names>
</name>
<name>
<surname>Nicholas</surname> <given-names>J</given-names>
</name>
</person-group>. <source>The Accuracy of Chest Ct in the Diagnosis of Covid-19: An Umbrella Review. Website</source> (<year>2021</year>). Available at: <uri xlink:href="https://www.cebm.net/covid-19/the-accuracy-of-chest-ct-in-the-diagnosis-of-covid-19-an-umbrella-review/">https://www.cebm.net/covid-19/the-accuracy-of-chest-ct-in-the-diagnosis-of-covid-19-an-umbrella-review/</uri>.</citation>
</ref>
<ref id="B10">
<label>10</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nambu</surname> <given-names>A</given-names>
</name>
<name>
<surname>Ozawa</surname> <given-names>K</given-names>
</name>
<name>
<surname>Kobayashi</surname> <given-names>N</given-names>
</name>
<name>
<surname>Tago</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Imaging of Community-Acquired Pneumonia: Roles of Imaging Examinations, Imaging Diagnosis of Specific Pathogens and Discrimination From Noninfectious Diseases</article-title>. <source>World J Radiol</source> (<year>2014</year>) <volume>6</volume>:<fpage>779</fpage>. doi: <pub-id pub-id-type="doi">10.4329/wjr.v6.i10.779</pub-id>
</citation>
</ref>
<ref id="B11">
<label>11</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wootton</surname> <given-names>D</given-names>
</name>
<name>
<surname>Feldman</surname> <given-names>C</given-names>
</name>
</person-group>. <article-title>The Diagnosis of Pneumonia Requires a Chest Radiograph (X-Ray)&#x2014;Yes, No or Sometimes</article-title>? <source>Pneumonia</source> (<year>2014</year>) <volume>5</volume>:<fpage>1</fpage>&#x2013;<lpage>7</lpage>. doi: <pub-id pub-id-type="doi">10.15172/pneu.2014.5/464</pub-id>
</citation>
</ref>
<ref id="B12">
<label>12</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname> <given-names>F</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Z</given-names>
</name>
<etal/>
</person-group>. <article-title>Review of Artificial Intelligence Techniques in Imaging Data Acquisition, Segmentation, and Diagnosis for Covid-19</article-title>. <source>IEEE Rev Biomed Eng</source> (<year>2020</year>) <volume>14</volume>:<fpage>4</fpage>&#x2013;<lpage>15</lpage>. doi: <pub-id pub-id-type="doi">10.1109/RBME.2020.2987975</pub-id>
</citation>
</ref>
<ref id="B13">
<label>13</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dong</surname> <given-names>D</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S</given-names>
</name>
<name>
<surname>Hui</surname> <given-names>H</given-names>
</name>
<name>
<surname>Gong</surname> <given-names>L</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>The Role of Imaging in the Detection and Management of Covid-19: A Review</article-title>. <source>IEEE Rev Biomed Eng</source> (<year>2020</year>) <volume>14</volume>:<fpage>16</fpage>&#x2013;<lpage>29</lpage>. doi: <pub-id pub-id-type="doi">10.1109/RBME.2020.2990959</pub-id>
</citation>
</ref>
<ref id="B14">
<label>14</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ouyang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Huo</surname> <given-names>J</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>L</given-names>
</name>
<name>
<surname>Shan</surname> <given-names>F</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Mo</surname> <given-names>Z</given-names>
</name>
<etal/>
</person-group>. <article-title>Dual-Sampling Attention Network for Diagnosis of Covid-19 From Community Acquired Pneumonia</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2595&#x2013;605</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2995508</pub-id>
</citation>
</ref>
<ref id="B15">
<label>15</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Shi</surname> <given-names>F</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>L</given-names>
</name>
<name>
<surname>Shan</surname> <given-names>F</given-names>
</name>
<name>
<surname>Song</surname> <given-names>B</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>D</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Large-Scale Screening to Distinguish Between Covid-19 and Community-Acquired Pneumonia Using Infection Size-Aware Classification</article-title>. <source>Phys Med Biol</source> (<year>2021</year>) <volume>66</volume>:<fpage>065031</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1088/1361-6560/abe838</pub-id>
</citation>
</ref>
<ref id="B16">
<label>16</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fan</surname> <given-names>D</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>T</given-names>
</name>
<name>
<surname>Ji</surname> <given-names>G</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>G</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Inf-Net: Automatic Covid-19 Lung Infection Segmentation From Ct Images</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2626&#x2013;37</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2996645</pub-id>
</citation>
</ref>
<ref id="B17">
<label>17</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kermany</surname> <given-names>DS</given-names>
</name>
<name>
<surname>Goldbaum</surname> <given-names>M</given-names>
</name>
<name>
<surname>Cai</surname> <given-names>W</given-names>
</name>
<name>
<surname>Valentim</surname> <given-names>CCS</given-names>
</name>
<name>
<surname>Liang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Baxter</surname> <given-names>SL</given-names>
</name>
<etal/>
</person-group>. <article-title>Identifying Medical Diagnoses and Treatable Diseases by Image-Based Deep Learning</article-title>. <source>cell</source> (<year>2018</year>) <volume>172</volume>:<page-range>1122&#x2013;31</page-range>. doi: <pub-id pub-id-type="doi">10.1016/j.cell.2018.02.010</pub-id>
</citation>
</ref>
<ref id="B18">
<label>18</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rajaraman</surname> <given-names>S</given-names>
</name>
<name>
<surname>Candemir</surname> <given-names>S</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>I</given-names>
</name>
<name>
<surname>Thoma</surname> <given-names>G</given-names>
</name>
<name>
<surname>Antani</surname> <given-names>S</given-names>
</name>
</person-group>. <article-title>Visualization and Interpretation of Convolutional Neural Network Predictions in Detecting Pneumonia in Pediatric Chest Radiographs</article-title>. <source>Appl Sci</source> (<year>2018</year>) <volume>8</volume>:<fpage>1715</fpage>. doi: <pub-id pub-id-type="doi">10.3390/app8101715</pub-id>
</citation>
</ref>
<ref id="B19">
<label>19</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Han</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>B</given-names>
</name>
<name>
<surname>Hong</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Li</surname> <given-names>T</given-names>
</name>
<name>
<surname>Cong</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>X</given-names>
</name>
<etal/>
</person-group>. <article-title>Accurate Screening of Covid-19 Using Attention-Based Deep 3d Multiple Instance Learning</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2584&#x2013;94</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2996256</pub-id>
</citation>
</ref>
<ref id="B20">
<label>20</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kohno</surname> <given-names>N</given-names>
</name>
<name>
<surname>Ikezoe</surname> <given-names>J</given-names>
</name>
<name>
<surname>Johkoh</surname> <given-names>T</given-names>
</name>
<name>
<surname>Takeuchi</surname> <given-names>N</given-names>
</name>
<name>
<surname>Tomiyama</surname> <given-names>N</given-names>
</name>
<name>
<surname>Kido</surname> <given-names>S</given-names>
</name>
<etal/>
</person-group>. <article-title>Focal Organizing Pneumonia: CT Appearance</article-title>. <source>Radiology</source> (<year>1993</year>) <volume>189</volume>:<page-range>119&#x2013;23</page-range>. doi: <pub-id pub-id-type="doi">10.1148/radiology.189.1.8372180</pub-id>
</citation>
</ref>
<ref id="B21">
<label>21</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>W</given-names>
</name>
<name>
<surname>Jacobs</surname> <given-names>C</given-names>
</name>
<name>
<surname>Charbonnier</surname> <given-names>JP</given-names>
</name>
<name>
<surname>Van Ginneken</surname> <given-names>B</given-names>
</name>
</person-group>. <article-title>Relational Modeling for Robust and Efficient Pulmonary Lobe Segmentation in Ct Scans</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2664&#x2013;75</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2995108</pub-id>
</citation>
</ref>
<ref id="B22">
<label>22</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>G</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>X</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Ruan</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>A Noise-Robust Framework for Automatic Segmentation of Covid-19 Pneumonia Lesions From Ct Images</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2653&#x2013;63</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.3000314</pub-id>
</citation>
</ref>
<ref id="B23">
<label>23</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Narin</surname> <given-names>A</given-names>
</name>
<name>
<surname>Kaya</surname> <given-names>C</given-names>
</name>
<name>
<surname>Pamuk</surname> <given-names>Z</given-names>
</name>
</person-group>. <article-title>Automatic Detection of Coronavirus Disease (Covid-19) Using X-Ray Images and Deep Convolutional Neural Networks</article-title>. <source>Pattern Anal Appl</source> (<year>2021</year>) <volume>24</volume>:<page-range>1207&#x2013;20</page-range>. doi: <pub-id pub-id-type="doi">10.1007/s10044-021-00984-y</pub-id>
</citation>
</ref>
<ref id="B24">
<label>24</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bai</surname> <given-names>HX</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>R</given-names>
</name>
<name>
<surname>Xiong</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Hsieh</surname> <given-names>B</given-names>
</name>
<name>
<surname>Chang</surname> <given-names>K</given-names>
</name>
<name>
<surname>Halsey</surname> <given-names>K</given-names>
</name>
<etal/>
</person-group>. <article-title>AI Augmentation of Radiologist Performance in Distinguishing COVID-19 From Pneumonia of Other Etiology on Chest CT</article-title>. <source>Radiology</source> (<year>2020</year>) <volume>296</volume>:<page-range>E156&#x2013;65</page-range>. doi: <pub-id pub-id-type="doi">10.1148/radiol.2020201491</pub-id>
</citation>
</ref>
<ref id="B25">
<label>25</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Barstugan</surname> <given-names>M</given-names>
</name>
<name>
<surname>Ozkaya</surname> <given-names>U</given-names>
</name>
<name>
<surname>Ozturk</surname> <given-names>S</given-names>
</name>
</person-group>. <source>Coronavirus (Covid-19) Classification Using Ct Images by Machine Learning Methods</source>. <publisher-loc>Konya, Turkey</publisher-loc>: <publisher-name>arXiv preprint arXiv:2003.09424</publisher-name> (<year>2020</year>).</citation>
</ref>
<ref id="B26">
<label>26</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>L</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>L</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Kong</surname> <given-names>B</given-names>
</name>
<etal/>
</person-group>. <article-title>Artificial Intelligence Distinguishes Covid-19 From Community Acquired Pneumonia on Chest Ct</article-title>. <source>Radiology</source> (<year>2020</year>) <volume>0</volume>:<fpage>200905</fpage>. doi: <pub-id pub-id-type="doi">10.1148/radiol.2020200905</pub-id>
</citation>
</ref>
<ref id="B27">
<label>27</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hwang</surname> <given-names>EJ</given-names>
</name>
<name>
<surname>Park</surname> <given-names>S</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>KN</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>JI</given-names>
</name>
<name>
<surname>Choi</surname> <given-names>SY</given-names>
</name>
<name>
<surname>Lee</surname> <given-names>J</given-names>
</name>
<etal/>
</person-group>. <article-title>Development and Validation of a Deep Learning-Based Automated Detection Algorithm for Major Thoracic Diseases on Chest Radiographs</article-title>. <source>JAMA Network Open</source> (<year>2019</year>) <volume>2</volume>:<fpage>e191095</fpage>&#x2013;<lpage>e191095</lpage>. doi: <pub-id pub-id-type="doi">10.1001/jamanetworkopen.2019.1095</pub-id>
</citation>
</ref>
<ref id="B28">
<label>28</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>X</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>J</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>A Weakly-Supervised Framework for Covid-19 Classification and Lesion Localization From Chest Ct</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2615&#x2013;25</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2995965</pub-id>
</citation>
</ref>
<ref id="B29">
<label>29</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Roy</surname> <given-names>S</given-names>
</name>
<name>
<surname>Menapace</surname> <given-names>W</given-names>
</name>
<name>
<surname>Oei</surname> <given-names>S</given-names>
</name>
<name>
<surname>Luijten</surname> <given-names>B</given-names>
</name>
<name>
<surname>Fini</surname> <given-names>E</given-names>
</name>
<name>
<surname>Saltori</surname> <given-names>C</given-names>
</name>
<etal/>
</person-group>. <article-title>Deep Learning for Classification and Localization of Covid-19 Markers in Point-of-Care Lung Ultrasound</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2676&#x2013;87</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2994459</pub-id>
</citation>
</ref>
<ref id="B30">
<label>30</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Bao</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Lu</surname> <given-names>H</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>H</given-names>
</name>
<name>
<surname>Xiang</surname> <given-names>Y</given-names>
</name>
<etal/>
</person-group>. <article-title>Prior-Attention Residual Learning for More Discriminative Covid-19 Screening in Ct Images</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2572&#x2013;83</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2994908</pub-id>
</citation>
</ref>
<ref id="B31">
<label>31</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oh</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Park</surname> <given-names>S</given-names>
</name>
<name>
<surname>Ye</surname> <given-names>JC</given-names>
</name>
</person-group>. <article-title>Deep Learning Covid-19 Features on Cxr Using Limited Training Data Sets</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2688&#x2013;700</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2993291</pub-id>
</citation>
</ref>
<ref id="B32">
<label>32</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Alom</surname> <given-names>MZ</given-names>
</name>
<name>
<surname>Rahman</surname> <given-names>M</given-names>
</name>
<name>
<surname>Nasrin</surname> <given-names>MS</given-names>
</name>
<name>
<surname>Taha</surname> <given-names>TM</given-names>
</name>
<name>
<surname>Asari</surname> <given-names>VK</given-names>
</name>
</person-group>. <source>Covid_mtnet: Covid-19 Detection With Multi-Task Deep Learning Approaches</source>. <publisher-loc>Dayton, OH, USA</publisher-loc>: <publisher-name>arXiv preprint arXiv:2004.03747</publisher-name> (<year>2020</year>).</citation>
</ref>
<ref id="B33">
<label>33</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>&#xd6;zkaya</surname> <given-names>U</given-names>
</name>
<name>
<surname>&#xd6;zt&#xfc;rk</surname> <given-names>&#x15e;</given-names>
</name>
<name>
<surname>Barstugan</surname> <given-names>M</given-names>
</name>
</person-group>. <article-title>Coronavirus (Covid-19) Classification Using Deep Features Fusion and Ranking Technique</article-title>. In: <source>Big Data Analytics and Artificial Intelligence Against COVID-19: Innovation Vision and Approach</source>. <publisher-loc>Konya, Turkey</publisher-loc>: <publisher-name>Springer</publisher-name> (<year>2020</year>). p. <page-range>281&#x2013;95</page-range>.</citation>
</ref>
<ref id="B34">
<label>34</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>S</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>Z</given-names>
</name>
<etal/>
</person-group>. <article-title>Deep Learning Enables Accurate Diagnosis of Novel Coronavirus (Covid-19) With Ct Images</article-title>. <source>IEEE/ACM Trans Comput Biol Bioinf</source> (<year>2021</year>) <volume>34</volume>:<page-range>102&#x2013;6</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TCBB.2021.3065361</pub-id>
</citation>
</ref>
<ref id="B35">
<label>35</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Xia</surname> <given-names>L</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>F</given-names>
</name>
<name>
<surname>Wan</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Shi</surname> <given-names>F</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>H</given-names>
</name>
<etal/>
</person-group>. <article-title>Diagnosis of Coronavirus Disease 2019 (Covid-19) With Structured Latent Multi-View Representation Learning</article-title>. <source>IEEE Trans Med Imaging</source> (<year>2020</year>) <volume>39</volume>:<page-range>2606&#x2013;14</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2020.2992546</pub-id>
</citation>
</ref>
<ref id="B36">
<label>36</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Deep Residual Learning for Image Recognition</article-title>. <publisher-loc>Las Vegas, NV, USA</publisher-loc>: <publisher-name>CVPR</publisher-name> (<year>2016</year>). p. <page-range>770&#x2013;8</page-range>.</citation>
</ref>
<ref id="B37">
<label>37</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>CF</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>Q</given-names>
</name>
<name>
<surname>Mallinar</surname> <given-names>N</given-names>
</name>
<name>
<surname>Sercu</surname> <given-names>T</given-names>
</name>
<name>
<surname>Feris</surname> <given-names>R</given-names>
</name>
</person-group>. <source>Big-Little Net: An Efficient Multi-Scale Feature Representation for Visual and Speech Recognition</source>. <publisher-loc>New Orleans, USA</publisher-loc>: <publisher-name>arXiv preprint arXiv:1807.03848</publisher-name> (<year>2018</year>).</citation>
</ref>
<ref id="B38">
<label>38</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>H</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>B</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Kalantidis</surname> <given-names>Y</given-names>
</name>
<name>
<surname>Rohrbach</surname> <given-names>M</given-names>
</name>
<etal/>
</person-group>. <article-title>Drop an Octave: Reducing Spatial Redundancy in Convolutional Neural Networks With Octave Convolution</article-title>. <source>CVPR</source> (<year>2019</year>), <page-range>3435&#x2013;44</page-range>. doi: <pub-id pub-id-type="doi">10.1109/ICCV.2019.00353</pub-id>
</citation>
</ref>
<ref id="B39">
<label>39</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Cheng</surname> <given-names>B</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>R</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>T</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L</given-names>
</name>
</person-group>. <source>High Frequency Residual Learning for Multi-Scale Image Classification</source>. <publisher-loc>Cardiff, Wales</publisher-loc>: <publisher-name>arXiv preprint arXiv:1905.02649</publisher-name> (<year>2019</year>).</citation>
</ref>
<ref id="B40">
<label>40</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname> <given-names>S</given-names>
</name>
<name>
<surname>Cheng</surname> <given-names>M</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>K</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>M</given-names>
</name>
<name>
<surname>Torr</surname> <given-names>P</given-names>
</name>
</person-group>. <article-title>Res2net: A New Multi-Scale Backbone Architecture</article-title>. <source>IEEE Trans Pattern Anal Mach Intell</source> (<year>2021</year>) <volume>43</volume>:<page-range>652&#x2013;62</page-range>. doi: <pub-id pub-id-type="doi">10.1109/TPAMI.2019.2938758</pub-id>
</citation>
</ref>
<ref id="B41">
<label>41</label>
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hochreiter</surname> <given-names>S</given-names>
</name>
<name>
<surname>Schmidhuber</surname> <given-names>J</given-names>
</name>
</person-group>. <article-title>Long Short-Term Memory</article-title>. <source>Neural Comput</source> (<year>1997</year>) <volume>9</volume>:<page-range>1735&#x2013;80</page-range>. doi: <pub-id pub-id-type="doi">10.1162/neco.1997.9.8.1735</pub-id>
</citation>
</ref>
<ref id="B42">
<label>42</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Song</surname> <given-names>H</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>W</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>J</given-names>
</name>
<name>
<surname>Lam</surname> <given-names>KM</given-names>
</name>
</person-group>. <source>Pyramid Dilated Deeper Convlstm for Video Salient Object Detection</source>. <publisher-loc>Munich, Germany</publisher-loc>: <publisher-name>ECCV</publisher-name> (<year>2018</year>) p. <page-range>715&#x2013;31</page-range>.</citation>
</ref>
<ref id="B43">
<label>43</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Shi</surname> <given-names>X</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H</given-names>
</name>
<name>
<surname>Dit-Yan</surname>
</name>
<name>
<surname>Yeung</surname>
</name>
<name>
<surname>Wai-kin</surname> <given-names>W</given-names>
</name>
<etal/>
</person-group>. <article-title>Convolutional LSTM Network: A Machine Learning Approach for Precipitation Nowcasting</article-title>. In: <source>International Conference on Neural Information Processing Systems</source>. <publisher-loc>Montreal, Canada</publisher-loc>: <publisher-name>NIPS</publisher-name> (<year>2015</year>) p. <fpage>802</fpage>&#x2013;<lpage>C810</lpage>.</citation>
</ref>
<ref id="B44">
<label>44</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Selvaraju</surname> <given-names>RR</given-names>
</name>
<name>
<surname>Cogswell</surname> <given-names>M</given-names>
</name>
<name>
<surname>Das</surname> <given-names>A</given-names>
</name>
<name>
<surname>Vedantam</surname> <given-names>R</given-names>
</name>
<name>
<surname>Parikh</surname> <given-names>D</given-names>
</name>
<name>
<surname>Batra</surname> <given-names>D</given-names>
</name>
</person-group>. <source>Grad-Cam: Visual Explanations From Deep Networks via Gradient-Based Localization</source>. <publisher-loc>Venice, Italy</publisher-loc>: <publisher-name>ICCV</publisher-name> (<year>2017</year>) p. <page-range>618&#x2013;26</page-range>.</citation>
</ref>
<ref id="B45">
<label>45</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Zhu</surname> <given-names>J</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>N</given-names>
</name>
<name>
<surname>Xing</surname> <given-names>EP</given-names>
</name>
</person-group>. <source>Infinite Latent Svm for Classification and Multi-Task Learning</source>. <publisher-loc>Granada, Spain</publisher-loc>: <publisher-name>NIPS</publisher-name> (<year>2011</year>) p. <page-range>1620&#x2013;8</page-range>.</citation>
</ref>
<ref id="B46">
<label>46</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Krizhevsky</surname> <given-names>A</given-names>
</name>
<name>
<surname>Sutskever</surname> <given-names>I</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>GE</given-names>
</name>
</person-group>. <source>Imagenet Classification With Deep Convolutional Neural Networks</source>. <publisher-loc>Nevada, USA</publisher-loc>: <publisher-name>NIPS</publisher-name> (<year>2012</year>) p. <page-range>1097&#x2013;105</page-range>.</citation>
</ref>
<ref id="B47">
<label>47</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Simonyan</surname> <given-names>K</given-names>
</name>
<name>
<surname>Zisserman</surname> <given-names>A</given-names>
</name>
</person-group>. <source>Very Deep Convolutional Networks for Large-Scale Image Recognition</source>. <publisher-loc>San Diego, California, USA</publisher-loc>: <publisher-name>arXiv preprint arXiv:1409.1556</publisher-name> (<year>2014</year>).</citation>
</ref>
<ref id="B48">
<label>48</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Szegedy</surname> <given-names>C</given-names>
</name>
<name>
<surname>Vanhoucke</surname> <given-names>V</given-names>
</name>
<name>
<surname>Ioffe</surname> <given-names>S</given-names>
</name>
<name>
<surname>Shlens</surname> <given-names>J</given-names>
</name>
<name>
<surname>Wojna</surname> <given-names>Z</given-names>
</name>
</person-group>. <source>Rethinking the Inception Architecture for Computer Vision</source>. <publisher-loc>Las Vegas, NV, USA</publisher-loc>: <publisher-name>CVPR (IEEE</publisher-name> (<year>2016</year>) p. <page-range>2818&#x2013;26</page-range>.</citation>
</ref>
<ref id="B49">
<label>49</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Chollet</surname> <given-names>F</given-names>
</name>
</person-group>. <source>Xception: Deep Learning With Depthwise Separable Convolutions</source>. <publisher-loc>Honolulu, HI, USA</publisher-loc>: <publisher-name>CVPR</publisher-name> (<year>2017</year>) p. <page-range>1251&#x2013;8</page-range>.</citation>
</ref>
<ref id="B50">
<label>50</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Tran</surname> <given-names>D</given-names>
</name>
<name>
<surname>Bourdev</surname> <given-names>L</given-names>
</name>
<name>
<surname>Fergus</surname> <given-names>R</given-names>
</name>
<name>
<surname>Torresani</surname> <given-names>L</given-names>
</name>
<name>
<surname>Paluri</surname> <given-names>M</given-names>
</name>
</person-group>. <source>Learning Spatiotemporal Features With 3d Convolutional Networks</source>. <publisher-loc>Santiago, Chile</publisher-loc>: <publisher-name>ICCV</publisher-name> (<year>2015</year>) p. <page-range>4489&#x2013;97</page-range>.</citation>
</ref>
<ref id="B51">
<label>51</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Carreira</surname> <given-names>J</given-names>
</name>
<name>
<surname>Zisserman</surname> <given-names>A</given-names>
</name>
</person-group>. <source>Quo Vadis, Action Recognition? A New Model and the Kinetics Dataset</source>. <publisher-loc>Honolulu, HI, USA</publisher-loc>: <publisher-name>CVPR</publisher-name> (<year>2017</year>) p. <page-range>6299&#x2013;308</page-range>.</citation>
</ref>
<ref id="B52">
<label>52</label>
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>S</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>C</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J</given-names>
</name>
<name>
<surname>Tu</surname> <given-names>Z</given-names>
</name>
<name>
<surname>Murphy</surname> <given-names>K</given-names>
</name>
</person-group>. <source>Rethinking Spatiotemporal Feature Learning: Speed-Accuracy Trade-Offs in Video Classification</source>. <publisher-loc>Munich, Germany</publisher-loc>: <publisher-name>ECCV</publisher-name> (<year>2018</year>) p. <page-range>305&#x2013;21</page-range>.</citation>
</ref>
</ref-list>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>
<uri xlink:href="https://github.com/HzFu/COVID19">https://github.com/HzFu/COVID19.</uri>
</p>
</fn>
</fn-group>
</back>
</article>