<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Artif. Intell.</journal-id>
<journal-title>Frontiers in Artificial Intelligence</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Artif. Intell.</abbrev-journal-title>
<issn pub-type="epub">2624-8212</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/frai.2024.1467218</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Artificial Intelligence</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>SPEMix: a lightweight method via superclass pseudo-label and efficient mixup for echocardiogram view classification</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author"><name><surname>Ma</surname> <given-names>Shizhou</given-names></name><xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2796562/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Zhang</surname> <given-names>Yifeng</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2841233/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Li</surname> <given-names>Delong</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2790026/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Sun</surname> <given-names>Yixin</given-names></name><xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Qiu</surname> <given-names>Zhaowen</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/&#xD;&#xA;"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/&#xD;&#xA;"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author"><name><surname>Wei</surname> <given-names>Lei</given-names></name><xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/validation/&#xD;&#xA;"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/&#xD;&#xA;"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes"><name><surname>Dong</surname> <given-names>Suyu</given-names></name><xref ref-type="aff" rid="aff2"><sup>2</sup></xref><xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1498250/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>College of Aulin, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<aff id="aff2"><sup>2</sup><institution>College of Computer and Control Engineering, Northeast Forestry University</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<aff id="aff3"><sup>3</sup><institution>The Department of Ultrasound, Harbin Medical University Cancer Hospital</institution>, <addr-line>Harbin</addr-line>, <country>China</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Cardiovascular Surgery, First Affiliated Hospital With Nanjing Medical University</institution>, <addr-line>Nanjing</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Dongpo Xu, Northeast Normal University, China</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Sandeep Kumar Mishra, Yale University, United States</p>
<p>Aili Wang, Harbin University of Science and Technology, China</p>
<p>Junxiang Huang, Boston College, United States</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Suyu Dong, <email>dongsuyu@nefu.edu.cn</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>08</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>7</volume>
<elocation-id>1467218</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>07</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>25</day>
<month>11</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2025 Ma, Zhang, Li, Sun, Qiu, Wei and Dong.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Ma, Zhang, Li, Sun, Qiu, Wei and Dong</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec id="sec1001">
<title>Introduction</title>
<p>In clinical, the echocardiogram is the most widely used for diagnosing heart diseases. Different heart diseases are diagnosed based on different views of the echocardiogram images, so efficient echocardiogram view classification can help cardiologists diagnose heart disease rapidly. Echocardiogram view classification is mainly divided into supervised and semi-supervised methods. The supervised echocardiogram view classification methods have worse generalization performance due to the difficulty of labeling echocardiographic images, while the semi-supervised echocardiogram view classification can achieve acceptable results via a little labeled data. However, the current semi-supervised echocardiogram view classification faces challenges of declining accuracy due to out-of-distribution data and is constrained by complex model structures in clinical application.</p>
</sec>
<sec id="sec2001">
<title>Methods</title>
<p>To deal with the above challenges, we proposed a novel open-set semi-supervised method for echocardiogram view classification, SPEMix, which can improve performance and generalization by leveraging out-of-distribution unlabeled data. Our SPEMix consists of two core blocks, DAMix Block and SP Block. DAMix Block can generate a mixed mask that focuses on the valuable regions of echocardiograms at the pixel level to generate high-quality augmented echocardiograms for unlabeled data, improving classification accuracy. SP Block can generate a superclass pseudo-label of unlabeled data from the perspective of the superclass probability distribution, improving the classification generalization by leveraging the superclass pseudolabel.</p>
</sec>
<sec id="sec3001">
<title>Results</title>
<p>We also evaluate the generalization of our method on the Unity dataset and the CAMUS dataset. The lightweight model trained with SPEMix can achieve the best classification performance on the publicly available TMED2 dataset.</p>
</sec>
<sec id="sec4001">
<title>Discussion</title>
<p>For the first time, we applied the lightweight model to the echocardiogram view classification, which can solve the limits of the clinical application due to the complex model architecture and help cardiologists diagnose heart diseases more efficiently.</p>
</sec>
</abstract>
<kwd-group>
<kwd>superclass pseudo-label</kwd>
<kwd>lightweight</kwd>
<kwd>semi-supervised</kwd>
<kwd>open-set</kwd>
<kwd>echocardiogram view classification</kwd>
</kwd-group>
<counts>
<fig-count count="7"/>
<table-count count="5"/>
<equation-count count="20"/>
<ref-count count="41"/>
<page-count count="14"/>
<word-count count="9936"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Machine Learning and Artificial Intelligence</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>When diagnosing heart diseases based on the echocardiogram, different heart diseases depend on different views of the echocardiogram. For example, aortic stenosis can be diagnosed by analyzing the PLAX and PSAX views (<xref ref-type="bibr" rid="ref12">Huang et al., 2021</xref>), and early myocardial infarction can be detected via A4C and A2C (<xref ref-type="bibr" rid="ref5">Degerli et al., 2024</xref>). Hence, cardiologists usually need to identify the critical echocardiogram view during the clinical diagnostic process. Automated echocardiogram view classification can effectively reduce the clinical diagnosis time (<xref ref-type="bibr" rid="ref40">Zhu et al., 2022</xref>). <xref ref-type="fig" rid="fig1">Figure 1</xref> illustrates the clinical application process of the automated echocardiogram view classification. In practical clinical applications, automated view classification involved four steps. First, the cardiologist collected the echocardiogram data from the patient. Then, the cardiologist inputted the data into the automated view classification model to perform the echocardiogram view classification. Next, the cardiologist got the classification results of the echocardiogram. Finally, the cardiologist selected the expected view when making a diagnosis of a different disease and performed the diagnosis. Obviously, the core of automated echocardiogram view classification methods is to classify by the classification model. In this situation, developing a classification model that can identify the critical view precisely and efficiently will be vital in medicine. So many studies aim to develop echocardiogram view classification methods to assist cardiologists in diagnosing heart disease.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>The clinical application process of an automated echocardiogram view classification.</p>
</caption>
<graphic xlink:href="frai-07-1467218-g001.tif"/>
</fig>
<p>With the development of deep learning in medical images, there are many studies (<xref ref-type="bibr" rid="ref16">Kusunose et al., 2020</xref>; <xref ref-type="bibr" rid="ref24">Madani et al., 2018</xref>; <xref ref-type="bibr" rid="ref7">Gao et al., 2017</xref>) focused on the classification of echocardiogram views by deep learning. These methods can be divided into supervised methods and semi-supervised methods.</p>
<p>Supervised studies used a large number of labeled data to train a model. <xref ref-type="bibr" rid="ref24">Madani et al. (2018)</xref> performed the classification task by building a convolution model consisting of traditional convolution layers and fully connected layers. They classified 15 different echocardiogram views and got impressive classification results on their private dataset. <xref ref-type="bibr" rid="ref16">Kusunose et al. (2020)</xref> collected 17,000 echocardiogram images from 340 patients and built a neural network model. The model consisted of five convolution layers and five pooling layers to classify them into five categories. <xref ref-type="bibr" rid="ref7">Gao et al. (2017)</xref> fused the spatial and temporal information to perform eight view classifications based on their 432 video data. Although these supervised methods have gotten good results in identifying or detecting tasks (<xref ref-type="bibr" rid="ref7">Gao et al., 2017</xref>; <xref ref-type="bibr" rid="ref9">Han et al., 2024</xref>; <xref ref-type="bibr" rid="ref10">Han et al., 2024</xref>), they need massive labeled data. Unfortunately, it will take a lot of time and energy for medical experts to label the medical images. Furthermore, these studies (<xref ref-type="bibr" rid="ref16">Kusunose et al., 2020</xref>; <xref ref-type="bibr" rid="ref24">Madani et al., 2018</xref>; <xref ref-type="bibr" rid="ref7">Gao et al., 2017</xref>) used proprietary datasets, which made the generalization of their methods cannot be guaranteed. The private datasets also made it challenging to apply actual clinical needs. To address these issues, our study focused on the development of a semi-supervised algorithm within the publicly available dataset. We evaluated the generalization of our method in two different public datasets to prove the feasibility of our SPEMix in clinical application.</p>
<p>Semi-supervised studies (<xref ref-type="bibr" rid="ref25">Madani et al., 2018</xref>; <xref ref-type="bibr" rid="ref8">Hagberg et al., 2022</xref>) tried to decrease the cost of labeling data and improve efficiency by utilizing small labeled data and massive unlabeled data. Therefore, semi-supervised learning has been revealed to have superior potential (<xref ref-type="bibr" rid="ref35">Xiaojin, 2008</xref>), and it has also been widely used in medicine (<xref ref-type="bibr" rid="ref4">Chebli et al., 2018</xref>; <xref ref-type="bibr" rid="ref2">Bai et al., 2017</xref>). Many scholars (<xref ref-type="bibr" rid="ref25">Madani et al., 2018</xref>; <xref ref-type="bibr" rid="ref8">Hagberg et al., 2022</xref>; <xref ref-type="bibr" rid="ref14">Huang et al., 2023</xref>; <xref ref-type="bibr" rid="ref15">Huang et al., 2024</xref>) have used semi-supervised learning for the echocardiogram view classification task. <xref ref-type="bibr" rid="ref25">Madani et al. (2018)</xref> developed a semi-supervised generative adversarial network model to classify 15 views of echocardiogram via only a little labeled data. This method revealed the huge potential of semi-supervised. <xref ref-type="bibr" rid="ref8">Hagberg et al. (2022)</xref> innovatively developed a semi-supervised learning method using natural language processing for right ventricle view classification, which improved the efficiency of classification. Although these methods had good classification efficiency, their classification accuracy was usually worse than the classification accuracy of supervised methods due to ignoring the damage caused by the out-of-distribution data. Most current methods have not achieved accurate and reliable classification accuracy to meet practical clinical needs. We proposed a novel efficient mixup method for our semi-supervised method to solve the above problem. Different from other mixing methods (<xref ref-type="bibr" rid="ref38">Zhang et al., 2018</xref>; <xref ref-type="bibr" rid="ref36">Yun et al., 2019</xref>; <xref ref-type="bibr" rid="ref22">Liu et al., 2022</xref>) applied to natural images, our proposed efficient mixup method can automatically focus on the valuable pixel of the echocardiogram based on dynamic attention and produce high-quality augmented echocardiogram images. Specifically, we proposed a DAMix Block for our SPEMix to achieve our efficient augmentation.</p>
<p>Otherwise, many studies have observed that out-of-distribution of unlabeled data can harm the accuracy of semi-supervised learning (<xref ref-type="bibr" rid="ref26">Oliver et al., 2018</xref>; <xref ref-type="bibr" rid="ref21">Li et al., 2023</xref>; <xref ref-type="bibr" rid="ref39">Zhao et al., 2022</xref>). Out-of-distribution data means the data does not belong to any of the known classification categories. The semi-supervised methods achieved good results only when the unlabeled data shared the same class space with labeled data. Unfortunately, the collected unlabeled medical image datasets often included out-of-distribution data in practical medical applications. This phenomenon usually leads to unlabeled data sharing a different class space with labeled data. Some approaches tried to improve the medical image classification accuracy via adversarial attacks and defenses (<xref ref-type="bibr" rid="ref20">Lee et al., 2024</xref>; <xref ref-type="bibr" rid="ref17">Kwon, 2023</xref>; <xref ref-type="bibr" rid="ref18">Kwon and Lee, 2023</xref>). However, the dominant approaches to tackle the harm caused by the out-of-distribution data in natural images were to detect and filter the out-of-distribution data (<xref ref-type="bibr" rid="ref28">Saito et al., 2021</xref>; <xref ref-type="bibr" rid="ref3">Calderon-Ramirez et al., 2022</xref>). <xref ref-type="bibr" rid="ref14">Huang et al. (2023)</xref> used augmentation and step direction modification to reduce the harm of out-of-distribution for echocardiogram view classification. The proposed method utilized the gradient information from unlabeled data from out-of-distribution only when the out-of-distribution data could improve the model performance. Different from the previous techniques, filtering out the out-of-distribution or modifying the gradient descent updates, we designed a novel, effective open-set framework to make better use of the out-of-distribution unlabeled data. Inspired by the success of the superclass work (<xref ref-type="bibr" rid="ref21">Li et al., 2023</xref>) in solving the harm of out-of-distribution, we modeled the unlabeled data from superclass distribution. We assigned the superclass pseudo-label for unlabeled data. Specifically, we regarded all of the classes out of the distribution as a new superclass, and we designed the SP Block to generate superclass pseudo-labels to leverage the information in unlabeled datasets better. The SP Block included a multiclass classifier and a close-set classifier to calculate the out-of-distribution class probability distribution and the in-the-distribution class probability distribution. SP Block can calculate the superclass probability distribution and generate the superclass pseudo-labels based on these probabilities. Then, our superclass pseudo-labels can be leveraged by an open-set classifier, and our model can learn the semantic features of unlabeled data that are out of the distribution.</p>
<p>On the other hand, some researcher (<xref ref-type="bibr" rid="ref1">Avola et al., 2024</xref>) focused on developing more complex models to improve the accuracy of classification. <xref ref-type="bibr" rid="ref1">Avola et al. (2024)</xref> built a multi-scale feature transformer model to capture different scale details to get better classification results in the TMED2 (<xref ref-type="bibr" rid="ref13">Huang et al., 2022</xref>) dataset. Although this model can provide multi-view and single-view recognition, the transformer model has a more significant number of parameters and lower computational efficiency compared with the lightweight models. To satisfy real medical clinical needs, for the first time, we applied the lightweight model to echocardiogram view classification to improve the classification efficiency. RepViT (<xref ref-type="bibr" rid="ref34">Wang et al., 2024</xref>) model designed a novel lightweight CNN based on the structures of lightweight VIT and achieved the best performance of the lightweight model. Inspired by the great work, we designed a four-layer lightweight encoder based on the RepViT for view classification. To improve our training efficiency, we decoupled the mixed data generate stage and the pseudo-label generate stage. Specifically, we embedded the DAMix Block into the teacher encoder to provide high-quality mixed data to the classifier. We embedded the SP Block into the student model to leverage the out-of-distribution.</p>
<p>In this paper, we proposed a lightweight open-set semi-supervised method, SPEMix, for echocardiogram view classification. Different from traditional semi-supervised learning classification algorithms, such as FixMatch and OpenMatch, our proposed SPEMix can classify the echocardiogram views accurately and efficiently, providing a new perspective on practical clinical application. FixMatch achieves semi-supervised learning by assigning pseudo-labels to unlabeled data but discards unlabeled data with low thresholds, making it difficult to consider unseen classes. In contrast, our SPEMix assigns open-set pseudo-labels to each unlabeled data point, leveraging the semantic information of each unlabeled instance. Compared to OpenMatch, our open-set semi-supervised method introduces mixup data augmentation, which improves classification accuracy through joint supervision of unseen classes and mixed data. In summary, our contributions are as follows:</p>
<list list-type="order">
<list-item>
<p>We innovatively proposed a mixed data generator (DAMix) which introduced the dynamic attention mechanism for the medical echocardiogram. The DAMix Block can provide the mixed mask embedded with the mixing ratio. The mixed mask, which consists of valuable information, can mix up the unlabeled echocardiogram efficiently at the pixel level.</p>
</list-item>
<list-item>
<p>We proposed a novel method to leverage the out-of-distribution unlabeled data from superclass probability contribution. We designed an SP Block to model the unlabeled data in terms of superclass distribution to generate the pseudo-labels of unlabeled data. Then, we introduced an open-set classifier to calculate the superclass prediction. We also proposed a novel open-set loss based on consistent regularization to utilize the superclass pseudo-label.</p>
</list-item>
<list-item>
<p>Finally, we applied the lightweight model to our semi-supervised method and built a lightweight encoder based on RepViT, which improved the accuracy and efficiency of our proposed method.</p>
</list-item>
</list>
</sec>
<sec sec-type="methods" id="sec2">
<label>2</label>
<title>Methods</title>
<p>In this section, we introduced the detailed information of the proposed novel framework, SPEMix, for echocardiogram view classification. The framework of the proposed SPEMix was shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>. The two core components of the SPEMix, DAMix Block and SP Block, can, respectively, generate the mixed mask to mix up data and superclass pseudo-label to leverage the out-of-distribution data. Meanwhile, the SPEMix used the lightweight encoder model to improve efficiency. The lightweight teacher model consisted of the DAMix Block to generate the augmentation data. The lightweight student model consisted of the SP Block to generate the superclass pseudo-label. The teacher model and student model were trained in an end-to-end manner to enable data enhancement and superclass pseudo-labels generation to work together.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>The framework of the proposed method <bold>(A)</bold>. Specifically, the DAMix Block was embedded behind the third layer of the teacher model and generated the mixup mask to mix up the input of the unlabeled data. The SP Block was added at the end of the student model to generate the superclass pseudo-label. The open-set classifier can calculate the superclass prediction. The legend of the framework <bold>(A)</bold> is shown in <bold>(B)</bold>. The process of updating SPEMix is shown in <bold>(C)</bold>. The parameters of the student model were updated via backpropagation, and the teacher parameters were updated by the exponential moving average strategy of the student model parameters.</p>
</caption>
<graphic xlink:href="frai-07-1467218-g002.tif"/>
</fig>
<sec id="sec3">
<label>2.1</label>
<title>DAMix block for the generation of high-quality augmented echocardiograms</title>
<p>Mixing augmentation technology (<xref ref-type="bibr" rid="ref38">Zhang et al., 2018</xref>) is useful for improving the accuracy and generalization of classification tasks. However, the traditional mixing augmentation methods did not work well for echocardiogram view classification because most echocardiograms are grayscale images (<xref ref-type="bibr" rid="ref31">Shorten and Khoshgoftaar, 2019</xref>). To address these issues and improve the accuracy and generalization of echocardiogram view classification, we proposed DAMix Block to perform pixel-level Efficient Mixup. The DAMix Block generated the mixed mask <inline-formula>
<mml:math id="M1">
<mml:mi mathvariant="normal">M</mml:mi>
</mml:math>
</inline-formula> that contains crucial information for performing the pixel-level Efficient Mixup process. This DAMix Block was capable of embedding the mixing ratio into the mixed mask. The framework of the DAMix Block was shown in <xref ref-type="fig" rid="fig3">Figure 3</xref>.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>The overall framework of the DAMix Block is shown in <bold>(A)</bold>. The structure of the Value Block, Key Block and Query Block is shown in <bold>(B)</bold>.</p>
</caption>
<graphic xlink:href="frai-07-1467218-g003.tif"/>
</fig>
<p>Inspired by the improvement of the sparse self-attention in the vision transformer (<xref ref-type="bibr" rid="ref41">Zhu et al., 2023</xref>), the DAMix Block first introduced the dynamic attention mechanism for the echocardiogram. The dynamic attention mechanism divided the echocardiogram into sub-regions and searched for the regions containing critical information in the feature map. This process helped the generated mask to capture the global features of the entire image efficiently. The process of the dynamic attention mechanism was as follows: Initially, the dynamic attention mechanism divided the echocardiogram feature <inline-formula>
<mml:math id="M2">
<mml:mi>x</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> into <inline-formula>
<mml:math id="M3">
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:math>
</inline-formula> different small square regions, so the size of each region will be <inline-formula>
<mml:math id="M4">
<mml:mfrac>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mfrac>
</mml:math>
</inline-formula>. We regarded each of the square regions as a token. Each token had <inline-formula>
<mml:math id="M5">
<mml:mfrac>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mfrac>
</mml:math>
</inline-formula> features and there were <inline-formula>
<mml:math id="M6">
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:math>
</inline-formula>tokens in total. This process can be achieved by reshaping the feature map into <inline-formula>
<mml:math id="M7">
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">H</mml:mi>
<mml:mi mathvariant="normal">W</mml:mi>
</mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mspace width="thickmathspace"/>
<mml:mtext>.</mml:mtext>
</mml:math>
</inline-formula>&#x202F;After the divided step, we will get the echocardiogram token features <inline-formula>
<mml:math id="M8">
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula>. Then, the dynamic attention mechanism calculated the region Query(Q), region Key(K), and region Value(V) of the token features <inline-formula>
<mml:math id="M9">
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula> by linear projections. The process is shown in <xref ref-type="disp-formula" rid="EQ2">Equation 1</xref>:</p>
<disp-formula id="EQ2">
<label>(1)</label>
<mml:math id="M10">
<mml:mi mathvariant="normal">Q</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:msup>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">q</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">K</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:msup>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msup>
<mml:mo>,</mml:mo>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:msup>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msup>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M11">
<mml:msup>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">q</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> was the linear projection weight of the region query, <inline-formula>
<mml:math id="M12">
<mml:msup>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> was the linear projection weight of the region key, and <inline-formula>
<mml:math id="M13">
<mml:msup>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">v</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> was the linear projection weight of the region value.</p>
<p>Furthermore, the dynamic attention mechanism calculated the average matrix of the region Query Q, and region Key <inline-formula>
<mml:math id="M14">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>. Specifically, this dynamic attention can calculate the average feature of each token. This process can be achieved by performing 2D average pooling with a kernel size of <inline-formula>
<mml:math id="M15">
<mml:mfrac>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>W</mml:mi>
</mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mfrac>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:math>
</inline-formula> and reshaping the pooling results into <inline-formula>
<mml:math id="M16">
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
</mml:math>
</inline-formula>C as the average matrix <inline-formula>
<mml:math id="M17">
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mi>a</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M18">
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>a</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>.</p>
<p>To search for the crucial regions of the echocardiogram efficiently, the similarity between each region was calculated by the result of the dot product of the average matrix of the query and key. The process of calculating the similarity is shown in <xref ref-type="disp-formula" rid="EQ2">Equation 2</xref>:</p>
<disp-formula id="EQ3">
<label>(2)</label>
<mml:math id="M19">
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mi>a</mml:mi>
</mml:msup>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>a</mml:mi>
</mml:msup>
</mml:mfenced>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:msup>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M20">
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> denoted the similarity of the regions; <inline-formula>
<mml:math id="M21">
<mml:msup>
<mml:mi>Q</mml:mi>
<mml:mi>a</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> denoted the average matrix of the Q, <inline-formula>
<mml:math id="M22">
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>a</mml:mi>
</mml:msup>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:msup>
<mml:mi>S</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula> denoted the average matrix of the <inline-formula>
<mml:math id="M23">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula>. The last step of the dynamic-attention mechanism was to find the index <inline-formula>
<mml:math id="M24">
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>i</mml:mi>
<mml:msub>
<mml:mi>d</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</inline-formula> of the top k most relevant tokens for each token, then we collected all of the relevant tokens of <inline-formula>
<mml:math id="M25">
<mml:mi>K</mml:mi>
</mml:math>
</inline-formula> and V as a new <inline-formula>
<mml:math id="M26">
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>c</mml:mi>
</mml:msup>
</mml:math>
</inline-formula> and a new <inline-formula>
<mml:math id="M27">
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>c</mml:mi>
</mml:msup>
</mml:math>
</inline-formula>, which can be achieved by searching for the index of <inline-formula>
<mml:math id="M28">
<mml:mi mathvariant="normal">P</mml:mi>
</mml:math>
</inline-formula>. The dynamic attention can be represented as the <xref ref-type="disp-formula" rid="EQ1">Equation 3</xref>:</p>
<disp-formula id="EQ1">
<label>(3)</label>
<mml:math id="M29">
<mml:mi mathvariant="italic">output</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">softmax</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mfrac>
<mml:mrow>
<mml:mi>Q</mml:mi>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:msup>
<mml:mi>K</mml:mi>
<mml:mi>c</mml:mi>
</mml:msup>
</mml:mfenced>
<mml:mi>T</mml:mi>
</mml:msup>
</mml:mrow>
<mml:msqrt>
<mml:mi>c</mml:mi>
</mml:msqrt>
</mml:mfrac>
</mml:mfenced>
<mml:msup>
<mml:mi>V</mml:mi>
<mml:mi>c</mml:mi>
</mml:msup>
</mml:math>
</disp-formula>
<p>DAMix Block was guided by the dynamic attention mechanism to find regions in the echocardiogram that are useful for the classification task. In addition, DAMix Block embedded the mixing rates into the corresponding feature maps and used the idea of cross-attention to generate appropriate mixed masks for the echocardiogram. The process of mixed mask generation can be formulated as follows: getting an unlabeled image pair <inline-formula>
<mml:math id="M30">
<mml:mfenced open="(" close=")" separators=",">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mfenced>
</mml:math>
</inline-formula> from the minibatch of the unlabeled dataset. The feature maps from the k-th layer of the unlabeled image pair were inputted into the DAMix Block. The dynamic attention module was subsequently employed to calculate the weighted feature maps <inline-formula>
<mml:math id="M31">
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
<mml:mo>,</mml:mo>
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:math>
</inline-formula>. Furthermore, to achieve the pixel-level mixup, we designed the Value Block, Query Block, and Key Block to calculate the Value matrix embedded with <italic>&#x03BB;</italic>, Query matrix embedded with &#x03BB;, and Key matrix embedded with &#x03BB; respectively, where &#x03BB;<inline-formula>
<mml:math id="M32">
<mml:mo>&#x2208;</mml:mo>
<mml:mfenced open="[" close="]" separators=",">
<mml:mn>0</mml:mn>
<mml:mn>1</mml:mn>
</mml:mfenced>
</mml:math>
</inline-formula> is the mixing ratio that satisfies Beta distribution <inline-formula>
<mml:math id="M33">
<mml:mi>&#x03BB;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B2;</mml:mi>
<mml:mfenced open="(" close=")" separators=",">
<mml:mi>&#x03B1;</mml:mi>
<mml:mi>&#x03B1;</mml:mi>
</mml:mfenced>
</mml:math>
</inline-formula>. Specifically, these modules can splice a mixing rate matrix of the same size as the feature over the channel dimension of the feature, where each element of the mixing rate matrix was a randomly generated lambda <inline-formula>
<mml:math id="M34">
<mml:mi>&#x03BB;</mml:mi>
<mml:mtext>.</mml:mtext>
</mml:math>
</inline-formula>&#x202F;Then the DAMix Block got the Query matrix Q, Value matrix V, and Key matrix K, respectively, via the query block, value block, and key block. The structure of these blocks were shown in the right part of <xref ref-type="fig" rid="fig3">Figure 3</xref>. Notably, the Key block had the same structure as the Query block. To get the mixed mask <inline-formula>
<mml:math id="M35">
<mml:mi mathvariant="normal">M</mml:mi>
</mml:math>
</inline-formula>, we calculated the similarity matrix P of Q and K by the cross-attention mechanism, and generated the mixed mask <inline-formula>
<mml:math id="M36">
<mml:mi mathvariant="normal">M</mml:mi>
</mml:math>
</inline-formula> by using the similarity matrix <inline-formula>
<mml:math id="M37">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>and Value matrix V. The function of generating <inline-formula>
<mml:math id="M38">
<mml:mi mathvariant="normal">M</mml:mi>
</mml:math>
</inline-formula> can be formulated as the <xref ref-type="disp-formula" rid="EQ4">Equations 4</xref><xref ref-type="disp-formula" rid="EQ5">5</xref>:</p>
<disp-formula id="EQ4">
<label>(4)</label>
<mml:math id="M39">
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="italic">softmax</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mfrac>
<mml:mrow>
<mml:mi mathvariant="normal">K</mml:mi>
<mml:msup>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mfenced>
<mml:mi>T</mml:mi>
</mml:msup>
<mml:mo>&#x2297;</mml:mo>
<mml:mi>Q</mml:mi>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mn>2</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mfenced>
</mml:mrow>
<mml:mi mathvariant="normal">C</mml:mi>
</mml:mfrac>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ5">
<label>(5)</label>
<mml:math id="M40">
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi mathvariant="normal">U</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>&#x03C3;</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2297;</mml:mo>
<mml:mi mathvariant="normal">V</mml:mi>
<mml:mfenced open="(" close=")">
<mml:msubsup>
<mml:mi>z</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2032;</mml:mo>
</mml:msubsup>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M41">
<mml:mi>&#x03C3;</mml:mi>
</mml:math>
</inline-formula> was the Sigmoid activation function, V denoted the value block, K denoted the key block, Q denoted the query block, <inline-formula>
<mml:math id="M42">
<mml:mo>&#x2297;</mml:mo>
</mml:math>
</inline-formula> denoted the matrix multiplication, and C was a normalization factor, <inline-formula>
<mml:math id="M43">
<mml:mi mathvariant="normal">U</mml:mi>
</mml:math>
</inline-formula> denoted upsample function.</p>
<p>Based on the mixed masks generated by DAMix Block, we proposed the Efficient Mixup. The Efficient Mixup will generate high-quality mixed images via the mixed mask. Efficient Mixup achieved linear interpolation through pixel-wise multiplication between data and masks, generating high-quality mixed echocardiograms. When performing the efficient mixup between <inline-formula>
<mml:math id="M44">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M45">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>, we regard unlabeled image <inline-formula>
<mml:math id="M46">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> as the value. The DAMix Block calculated the mixed mask M for the value image. Similarly, the mixed mask of the value image <inline-formula>
<mml:math id="M47">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>was 1-M generated via the DAMix Block. The specific formula for efficient mixup was the <xref ref-type="disp-formula" rid="EQ6">Equation 6</xref>:</p>
<disp-formula id="EQ6">
<label>(6)</label>
<mml:math id="M48">
<mml:mi mathvariant="normal">Efficient Mixup</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>M</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2299;</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M49">
<mml:mo>&#x2299;</mml:mo>
</mml:math>
</inline-formula> denoted element-wise product, <inline-formula>
<mml:math id="M50">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mi mathvariant="italic">mix</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> was the output result of the DAMix Block. The visualization results of the mixed masks generated via DAMix Block and visualization results of mixup data generated via Efficient Mixup were shown in the <xref ref-type="fig" rid="fig4">Figure 4</xref>.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p>Visualization results of the mixed masks generated via DAMix Block and visualization results of mixup data generated via Efficient Mixup.</p>
</caption>
<graphic xlink:href="frai-07-1467218-g004.tif"/>
</fig>
<p>In the training process of our SPEMix, following the previous successful work (<xref ref-type="bibr" rid="ref22">Liu et al., 2022</xref>), we also embedded our DAMix Block behind the third layer of the encoder. The DAMix loss function can update the parameters of the DAMix Block, a combination of the unlabeled augmentation data generation loss and the labeled data classification loss. To calculate the augmentation generation loss, we proposed the novel generation loss based on the mixed cross-entropy loss function (<xref ref-type="bibr" rid="ref22">Liu et al., 2022</xref>) for the mixed unlabeled augmentation data. In the previous work, the cross entropy loss between mixed prediction and the mixup of labels was represented by the mixed cross-entropy loss function. Hence, we calculate our mixed generation loss by modifying the mixed cross-entropy. The view of our unlabeled mixed cross-entropy loss function was as follows: we considered the prediction <inline-formula>
<mml:math id="M51">
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> of unlabeled data <inline-formula>
<mml:math id="M52">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> before augmentation as the label of the unlabeled data. The generation loss function can be formulated as the <xref ref-type="disp-formula" rid="EQ7">Equation 7</xref>:</p>
<disp-formula id="EQ7">
<label>(7)</label>
<mml:math id="M53">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x03BB;</mml:mi>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")" separators=",">
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi mathvariant="italic">mix</mml:mi>
</mml:msub>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03BB;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")" separators=",">
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi mathvariant="italic">mix</mml:mi>
</mml:msub>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M54">
<mml:mi>&#x03BB;</mml:mi>
</mml:math>
</inline-formula>meant the mixing ratio, and <inline-formula>
<mml:math id="M55">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>meant the cross-entropy loss function.</p>
<p>However, the encoder&#x2019;s predictions of the unlabeled data before augmentation were unreliable in the early stages of training, which leaded to the wrong optimization of the generation loss. To address this issue, we also used the supervision of labeled data to assist in the generation of the mixup mask to guide the process of optimizing the DAMix Block. The process can be achieved by the cross-entropy of the labeled data, the process can be formulated as the <xref ref-type="disp-formula" rid="EQ8">Equation 8</xref>:</p>
<disp-formula id="EQ8">
<label>(8)</label>
<mml:math id="M56">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">labeled</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")" separators=",">
<mml:mi>y</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>where y denoted the ground truth of the labeled data, and <inline-formula>
<mml:math id="M57">
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> denoted the prediction of the labeled data. Finally, the loss of the DAMix Block denoted the combination of the generation loss and the cross-entropy loss of the labeled data, be formulated as the <xref ref-type="disp-formula" rid="EQ9">Equation 9</xref>:<disp-formula id="EQ9">
<label>(9)</label>
<mml:math id="M58">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">DAMix</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">labeled</mml:mi>
</mml:msub>
</mml:math>
</disp-formula></p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>SP block for the generation of the echocardiogram superclass pseudo-label</title>
<p>To utilize the unlabeled data of out-of-distribution, we proposed a novel superclass pseudo-label from the perspective of superclass probability. Specifically, our approach first considered all in-the-distribution classes as the in-the-distribution superclass and all out-of-distribution class data as an out-of-distribution superclass. Then, we modeled the superclass from the perspective of the probability distribution to generate the superclass pseudo-label.</p>
<p>We designed the SP Block(short for Superclass Pseudo-label generator) to achieve the above process. The framework of the SP Block is shown in <xref ref-type="fig" rid="fig5">Figure 5</xref>. The SP Block included a multiclass classifier to calculate the out-of-distribution class probability distribution, a close-set classifier to calculate the in-the-distribution class probability distribution, and a SP Generator to get the superclass pseudo-label. The black column of P in <xref ref-type="fig" rid="fig5">Figure 5</xref> represented the probability of the sample belonging to each in-the-distribution class only considering the presence of the in-the-distribution classes. The blue columns of Q in <xref ref-type="fig" rid="fig5">Figure 5</xref> represented the probability of the sample belonging to each in-the-distribution class accounting for the unknown classes. Notably, the black column of P was not the same as the blue column of Q. While the orange columns of Q in <xref ref-type="fig" rid="fig5">Figure 5</xref> represented the probability of the sample not belonging to each in-the-distribution class.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>The framework of the SP Block. Specifically, the multiclass classifier calculated the probability distribution of out-of-distribution class Q,&#x202F;<inline-formula>
<mml:math id="M59">
<mml:mi>Q</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mn>4</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mtext>.</mml:mtext>
</mml:math>
</inline-formula> The close-set classifier calculated the probability distribution of in-the-distribution class P,&#x202F;<inline-formula>
<mml:math id="M60">
<mml:mi>P</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mtext>.</mml:mtext>
</mml:math>
</inline-formula> Then the SP Generator acquired the superclass probability distribution and generated the Superclass Pseudo-label by assigning the weight matrix for the in-the-distribution superclass.</p>
</caption>
<graphic xlink:href="frai-07-1467218-g005.tif"/>
</fig>
<p>Our SP Block generated the superclass pseudo-label for unlabeled samples from the perspective of superclass probability distributions. Our SP Block acquired the in-the-distribution probability distribution P by calculating the probability that the sample belongs to each class in the distribution. Specifically, the close-set classifier, which is a normal fully connected layer, calculated the in-the-distribution probability P, P&#x220A;<inline-formula>
<mml:math id="M61">
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>. However, we can only get the true class prediction when this unlabeled sample belongs to the in-the-distribution classes. Furthermore, we calculated the out-of-distribution probability distribution Q to model the out-of-distribution classes, which can be calculated by multi-binary classifiers. The multi-binary classifier has been proven to be able to detect whether a sample belongs to each in-the-distribution class,and the technology has been used widely in the previous open semi-supervised learning of filtering the out-of-distribution (<xref ref-type="bibr" rid="ref28">Saito et al., 2021</xref>). The multiclass classifier consisted of four binary classifiers and each binary classifier predicted the sample whether belonging to the k-th class. Specifically, <inline-formula>
<mml:math id="M62">
<mml:msub>
<mml:mi>&#x03A8;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> is a multiclass classifier consisted of k binary classifier <inline-formula>
<mml:math id="M63">
<mml:msub>
<mml:mi>&#x03A8;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
</mml:math>
</inline-formula>{<inline-formula>
<mml:math id="M64">
<mml:msub>
<mml:mi>&#x03C6;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x03C6;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x03C6;</mml:mi>
<mml:mn>3</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi>&#x03C6;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo stretchy="true">}</mml:mo>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math id="M65">
<mml:msub>
<mml:mi>&#x03C6;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> can output <inline-formula>
<mml:math id="M66">
<mml:mi mathvariant="normal">o</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi mathvariant="normal">q</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="normal">q</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mtext>,</mml:mtext>
</mml:math>
</inline-formula> where <inline-formula>
<mml:math id="M67">
<mml:msub>
<mml:mi mathvariant="normal">q</mml:mi>
<mml:mi mathvariant="normal">k</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> means the probability belongs to the k-th class of the sample. The output of <inline-formula>
<mml:math id="M68">
<mml:msub>
<mml:mi>&#x03A8;</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> was a matrix of k rows and 2 columns.</p>
<p>Then the SP Generator combined Q and P to generate the superclass pseudo-label, which can be achieved by the matrix multiplication of P and Q. The process of generating the superclass pseudo-label by the SP Generator had three steps: First, the SP Generator calculated the superclass probability distribution <inline-formula>
<mml:math id="M69">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")" separators=",">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="italic">in</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="italic">out</mml:mi>
</mml:msub>
</mml:mfenced>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211D;</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M70">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="italic">in</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> denoted the in-the-distribution superclass probability, <inline-formula>
<mml:math id="M71">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="italic">out</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> denoted the out-of-distribution superclass probability. Second, the SP Generator got the in-the-class probability weight matrix WM of the sample. Finally, the SP Generator assigned weights for the in-the-distribution superclass according to WM and got the superclass pseudo-label. The process of generating a superclass pseudo-label can be formulated as the process of <xref ref-type="disp-formula" rid="EQ10 EQ11 EQ12">Equations 10&#x2013;12</xref>:</p>
<disp-formula id="EQ10">
<label>(10)</label>
<mml:math id="M72">
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>P</mml:mi>
<mml:mo>&#x2297;</mml:mo>
<mml:mi>Q</mml:mi>
</mml:math>
</disp-formula>
<disp-formula id="EQ11">
<label>(11)</label>
<mml:math id="M73">
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03B1;</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x03B1;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:msub>
<mml:mi>&#x03B1;</mml:mi>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")" separators=",,,">
<mml:mfrac>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mfrac>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
<mml:mo>&#x2026;</mml:mo>
<mml:mfrac>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:msubsup>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi mathvariant="normal">i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:msubsup>
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ12">
<label>(12)</label>
<mml:math id="M74">
<mml:mi mathvariant="normal">S</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="italic">in</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi mathvariant="italic">out</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M75">
<mml:mi mathvariant="normal">W</mml:mi>
<mml:mi mathvariant="normal">M</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi mathvariant="normal">R</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x00D7;</mml:mo>
<mml:mi mathvariant="normal">n</mml:mi>
</mml:mrow>
</mml:msup>
</mml:math>
</inline-formula>, where n denoted the number of the in-the-distribution class. SP denoted the superclass pseudo-label. <inline-formula>
<mml:math id="M76">
<mml:msub>
<mml:mi mathvariant="normal">p</mml:mi>
<mml:mi mathvariant="normal">i</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> denoted the i-th element of the in-the-distribution probability P.</p>
<p>To optimize the SP Block, we, respectively, optimized the multiclass classifier and the close-set classifier with labeled data. For the loss of the multiclass classifier, we used the hard-negative sampling strategy (<xref ref-type="bibr" rid="ref29">Saito and Saenko, 2021</xref>), following the previous work. The loss function of the multiclass classifier can be formulated as the <xref ref-type="disp-formula" rid="EQ13">Equation 13</xref>:</p>
<disp-formula id="EQ13">
<label>(13)</label>
<mml:math id="M77">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>B</mml:mi>
</mml:mfrac>
<mml:munderover>
<mml:mstyle displaystyle="true">
<mml:mo stretchy="true">&#x2211;</mml:mo>
</mml:mstyle>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>B</mml:mi>
</mml:munderover>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mo>log</mml:mo>
<mml:mfenced open="(" close=")">
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mi mathvariant="normal">l</mml:mi>
<mml:mi>k</mml:mi>
<mml:mo>&#x2260;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>min</mml:mo>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mfenced open="(" close=")">
<mml:mover accent="true">
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Where B represented the batch size of the labeled data, <inline-formula>
<mml:math id="M78">
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> represented the first element in the y-th row of the out-of-distribution probability Q. <inline-formula>
<mml:math id="M79">
<mml:mover accent="true">
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>k</mml:mi>
</mml:msub>
<mml:mo stretchy="true">&#x00AF;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> represented the second element in the k-th row of the out-of-distribution probability Q.</p>
<p>For the loss of the close-set classifier, we used the cross-entropy loss function of the labeled data to optimize, was shown in the <xref ref-type="disp-formula" rid="EQ14">Equation 14</xref>:</p>
<disp-formula id="EQ14">
<label>(14)</label>
<mml:math id="M80">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">close</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")" separators=",">
<mml:mi>y</mml:mi>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>x</mml:mi>
</mml:msub>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ15">
<label>(15)</label>
<mml:math id="M81">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">close</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:math>
</disp-formula>
<p>The total loss of the SP Block was shown in <xref ref-type="disp-formula" rid="EQ15">Equation 15</xref>. In our SPEMix, we hoped the proposed DAMix Block and SP Block could work together to utilize the unlabeled datasets efficiently. Specifically, the SP Block can assign the superclass pseudo-label for augmentation data generated by the DAMix Block. To achieve the process, we built an open-set classifier to predict the superclass pseudo-label. We proposed the open-set loss function of SP Block based on the mixed process of the superclass pseudo-label to utilize the augmentation of unlabeled images. Specifically, the processes of calculating the open-set loss function are shown in the <xref ref-type="disp-formula" rid="EQ16 EQ17 EQ18">Equations 16&#x2013;18</xref>:</p>
<disp-formula id="EQ16">
<label>(16)</label>
<mml:math id="M82">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">op</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>&#x03BB;</mml:mi>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")" separators=",">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi mathvariant="italic">mix</mml:mi>
</mml:msub>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03BB;</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mfenced open="(" close=")" separators=",">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi mathvariant="italic">mix</mml:mi>
</mml:msub>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ17">
<label>(17)</label>
<mml:math id="M83">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>max</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<disp-formula id="EQ18">
<label>(18)</label>
<mml:math id="M84">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>max</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math id="M85">
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> meant the superclass pseudo-label of <inline-formula>
<mml:math id="M86">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>,<inline-formula>
<mml:math id="M87">
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> meant the superclass pseudo-label of <inline-formula>
<mml:math id="M88">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math id="M89">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> meant the corresponding class of the pseudo-label <inline-formula>
<mml:math id="M90">
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>. <inline-formula>
<mml:math id="M91">
<mml:msub>
<mml:mi mathvariant="normal">y</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula> meant the corresponding class of the pseudo-label <inline-formula>
<mml:math id="M92">
<mml:mi>S</mml:mi>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:math>
</inline-formula>. max was the function to caculate the index of the maximum probability. <inline-formula>
<mml:math id="M93">
<mml:msub>
<mml:mi>o</mml:mi>
<mml:mi mathvariant="italic">mix</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> denoted the superclass pseudo-label prediction of augmentation image from <inline-formula>
<mml:math id="M94">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M95">
<mml:msub>
<mml:mi>u</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:math>
</inline-formula> via the open-set classifier.</p>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>Lightweight encoder and end-to-end efficient learning paradigm</title>
<p>For the first time, we applied the lightweight model to the echocardiogram view classification. Specifically, we designed a lightweight encoder based on the RepViT (<xref ref-type="bibr" rid="ref34">Wang et al., 2024</xref>) network for our SPEMix. RepViT implemented the design of lightweight networks by using the structural re-parameterization (<xref ref-type="bibr" rid="ref6">Ding et al., 2019</xref>) principle. Hence, we built our lightweight encoder following the structure of RepViT to improve the efficiency of echocardiogram view classification. We built the lightweight student encoder and lightweight teacher encoder for our SPEMix. To satisfy our view classification tasks, our RepViT lightweight encoder only had four blocks.</p>
<p>Inspired by the success of AutoMix (<xref ref-type="bibr" rid="ref22">Liu et al., 2022</xref>), we also adopted the momentum update pipeline to decouple the process of augmentation of unlabeled images and the process of assigning the superclass pseudo-label. The architecture was shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>. We constructed two lightweight encoders with identical initialized parameters, which can help SPEMix synchronize the two processes by employing end-to-end training and achieve better accuracy and generalization. Specifically, the DAMix Block in the teacher model mixed up the unlabeled data, and the SP Block in the student model generated superclass pseudo labels for unlabeled data. The parameters of the student model can be updated by back propagation, while the teacher parameters were updated by the exponential moving average strategy (<xref ref-type="bibr" rid="ref27">Polyak and Juditsky, 1992</xref>) from the parameters of the student model. The total loss function of our SPEMix was the <xref ref-type="disp-formula" rid="EQ19">Equation 19</xref>:</p>
<disp-formula id="EQ19">
<label>(19)</label>
<mml:math id="M96">
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">total</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">DAMix</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mrow>
<mml:mi>S</mml:mi>
<mml:mi mathvariant="normal">P</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi mathvariant="script">&#x1D4C1;</mml:mi>
<mml:mi mathvariant="italic">op</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<p>The process of updating the teacher model can be reformulated as the <xref ref-type="disp-formula" rid="EQ20">Equation 20</xref>:</p>
<disp-formula id="EQ20">
<label>(20)</label>
<mml:math id="M97">
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">teacher</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>m</mml:mi>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">teacher</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mfenced open="(" close=")">
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">student</mml:mi>
</mml:msub>
</mml:math>
</disp-formula>
<p>where m was the momentum coefficient and<inline-formula>
<mml:math id="M98">
<mml:mi>m</mml:mi>
<mml:mi>&#x03F5;</mml:mi>
<mml:mo stretchy="true">(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="true">]</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">student</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> denoted the parameters of the student model, <inline-formula>
<mml:math id="M99">
<mml:msub>
<mml:mi>&#x03B8;</mml:mi>
<mml:mi mathvariant="italic">teacher</mml:mi>
</mml:msub>
</mml:math>
</inline-formula> denoted the parameters of teacher model.</p>
</sec>
</sec>
<sec sec-type="results" id="sec6">
<label>3</label>
<title>Results and discussion</title>
<sec id="sec7">
<label>3.1</label>
<title>Implement details</title>
<sec id="sec8">
<label>3.1.1</label>
<title>Dataset</title>
<p>We used the <italic>Tufts Medical Echocardiogram Dataset 2</italic> (TMED2) (<xref ref-type="bibr" rid="ref13">Huang et al., 2022</xref>) to train the proposed SPEMix and the CAMUS (<xref ref-type="bibr" rid="ref19">Leclerc et al., 2019</xref>) dataset and Unity (<xref ref-type="bibr" rid="ref11">Howard et al., 2021</xref>) dataset to evaluate the generalization. Specifically, the TMED2 contains four types of echocardiogram views, including PLAX, PLSX, A2C, and A4C. This dataset provides 353,500 unlabeled images and 24,964 labeled data collected from different patients. For TMED2, we used the officially released training sets, test sets, and validation sets. All of the resolution of this dataset is 112<inline-formula>
<mml:math id="M100">
<mml:mo>&#x00D7;</mml:mo>
</mml:math>
</inline-formula>112 pixels. We used the RandomCrop and RamdomHorizontalFlip as basic augments for the training dataset during training. The CAMUS (<xref ref-type="bibr" rid="ref19">Leclerc et al., 2019</xref>) dataset contains two view types including A2C, and A4C. We resized the resolution to 112<inline-formula>
<mml:math id="M101">
<mml:mo>&#x00D7;</mml:mo>
</mml:math>
</inline-formula>112 pixels to evaluate the generalization of the classification. The Unity (<xref ref-type="bibr" rid="ref11">Howard et al., 2021</xref>) dataset contains three view types, including PLAX, A2C, and A4C. We resized the resolution to 112<inline-formula>
<mml:math id="M102">
<mml:mo>&#x00D7;</mml:mo>
</mml:math>
</inline-formula>112 pixels to evaluate the generalization of the classification.</p>
</sec>
<sec id="sec9">
<label>3.1.2</label>
<title>Training setting</title>
<p>For training our SPEMix, we used our designed lightweight encoder with 112<inline-formula>
<mml:math id="M103">
<mml:mo>&#x00D7;</mml:mo>
</mml:math>
</inline-formula>112 size inputs. For the hyper-parameter of the SPEMix, the momentum coefficient was set to 0.999. For labeled data batch size and unlabeled data batch size, we set them to 64. We used the Adam optimizer to update the model parameters. For learning rate, we chose from the set of {0.1,0.01,0.001,0.0001}, and we reported the different best learning rate in different experiments. The learning rate of the SPEMix was set to 0.0001, we trained the SPEMix 500 epochs and adapted the learning rate by the Cosine Schedule (<xref ref-type="bibr" rid="ref23">Loshchilov and Hutter, 2022</xref>). We did not use the warm-up strategy. During the process of training, the parameters of the student model were updated via back propagation. And the parameters of the DAMix Block can update via the loss function of DAMix. The parameters of the DAMix Block will be frozen when the parameters of the teacher model update based on the parameters of student model. For comparison experiments, we used the Adam optimizers to train the other methods and we chose the best learning rate for each method. All of the comparison methods were trained on the TMED2. To make a fair comparison, all of our experiments were implemented on the GPU of the model NVIDIA A40.</p>
</sec>
</sec>
<sec id="sec10">
<label>3.2</label>
<title>Results of SPEMix</title>
<p>We reported the performance of our lightweight encoder via SPEMix on the TMED2 test dataset, the confusion matrix of the test dataset was shown in the left diagram of <xref ref-type="fig" rid="fig6">Figure 6</xref>. Each element of the matrix represented the probability of being predicted as the corresponding class, and the diagonal value represented the prediction accuracy of each view. The classification accuracy of the A2C view reached 96.97%, the classification accuracy of A4C reached 96.05%, the PLAX classification accuracy reached 98.29%, and the PSAX classification reached 96.61%. These results demonstrated that our SPEMix predicted every view very well. At the same time, we also provided the ROC curve to evaluate our classifier. We gave the ROC curve of SPEMix and calculated the AUC for each class. The larger values of the AUC mean the better the performance of our classifier.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>The evaluation result of the SPEMix. The left diagram has some problems due to decimal retention, so we provide a new confusion matrix on TMED2 <bold>(A)</bold> represents the confusion matrix of SPEMix on TMED2 and the right diagram <bold>(B)</bold> represents the ROC curve of TMED2.</p>
</caption>
<graphic xlink:href="frai-07-1467218-g006.tif"/>
</fig>
</sec>
<sec id="sec11">
<label>3.3</label>
<title>Comparison results with mixing methods</title>
<p>In this section, we compared our proposed SPEMix with the previous SOTA mixup methods to prove the advanced performance of our SPEMix, including CutMix (<xref ref-type="bibr" rid="ref36">Yun et al., 2019</xref>), SaliencyMix (<xref ref-type="bibr" rid="ref33">Uddin et al., 2006</xref>), and AutoMix (<xref ref-type="bibr" rid="ref22">Liu et al., 2022</xref>). CutMix represents the typical mixup method in natural images. SaliencyMix is the SOTA mixup that can generate mixed data by utilizing the saliency information of the natural images. AutoMix is the SOTA of the pixel-level mixup method which gets the most advanced performance in natural images. To compare these methods fairly, we, respectively, trained a common WideResNet encoder classifier and our lightweight encoder classifier by using each mixup method. We chose the best parameters for all of the methods in this comparison experiments. The best learning rate was set to 0.01. The batch size was set to 64, we all used the Cosine Schedule to adapt the learning rate and the training epoch was set to 500. All of the experiments did not use the warm-up strategy. The comparison result was shown in <xref ref-type="table" rid="tab1">Table 1</xref>. We reported the mean accuracy and standard deviation on the TMED2 from three different trails.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Comparison results between SPEMix and the previous Sota mixing methods. All results are expressed as percentages (%).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Method</th>
<th align="left" valign="top">Encoder</th>
<th align="center" valign="top">Test accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">CutMix (<xref ref-type="bibr" rid="ref36">Yun et al., 2019</xref>)</td>
<td align="left" valign="top">Wideresnet (<xref ref-type="bibr" rid="ref37">Zagoruyko and Komodakis, 2016</xref>)</td>
<td align="char" valign="top" char="&#x00B1;">94.30 &#x00B1; 0.2</td>
</tr>
<tr>
<td align="left" valign="top">CutMix (<xref ref-type="bibr" rid="ref36">Yun et al., 2019</xref>)</td>
<td align="left" valign="top">Lightweight encoder(Ours)</td>
<td align="char" valign="top" char="&#x00B1;">95.97 &#x00B1; 0.67</td>
</tr>
<tr>
<td align="left" valign="top">SaliencyMix (<xref ref-type="bibr" rid="ref33">Uddin et al., 2006</xref>)</td>
<td align="left" valign="top">Wideresnet (<xref ref-type="bibr" rid="ref37">Zagoruyko and Komodakis, 2016</xref>)</td>
<td align="char" valign="top" char="&#x00B1;">96.31 &#x00B1; 0.43</td>
</tr>
<tr>
<td align="left" valign="top">SaliencyMix (<xref ref-type="bibr" rid="ref33">Uddin et al., 2006</xref>)</td>
<td align="left" valign="top">Lightweight encoder(Ours)</td>
<td align="char" valign="top" char="&#x00B1;">96.87 &#x00B1; 0.2</td>
</tr>
<tr>
<td align="left" valign="top">AutoMix (<xref ref-type="bibr" rid="ref22">Liu et al., 2022</xref>)</td>
<td align="left" valign="top">Wideresnet (<xref ref-type="bibr" rid="ref37">Zagoruyko and Komodakis, 2016</xref>)</td>
<td align="char" valign="top" char="&#x00B1;">96.34 &#x00B1; 0.19</td>
</tr>
<tr>
<td align="left" valign="top">AutoMix (<xref ref-type="bibr" rid="ref22">Liu et al., 2022</xref>)</td>
<td align="left" valign="top">Lightweight encoder(Ours)</td>
<td align="char" valign="top" char="&#x00B1;">96.91 &#x00B1; 0.27</td>
</tr>
<tr>
<td align="left" valign="top">SPEMix(Ours)</td>
<td align="left" valign="top">Wideresnet (<xref ref-type="bibr" rid="ref37">Zagoruyko and Komodakis, 2016</xref>)</td>
<td align="char" valign="top" char="&#x00B1;"><bold>97.21 &#x00B1; 0.10</bold></td>
</tr>
<tr>
<td align="left" valign="top">SPEMix(Ours)</td>
<td align="left" valign="top">Lightweight encoder(Ours)</td>
<td align="char" valign="top" char="&#x00B1;"><bold>97.28 &#x00B1; 0.11</bold></td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>To make a fair comparison, we ran each method with three trials and reported the mean and standard deviation of test accuracy on the TMED2 dataset. The best performance of each encoder was highlighted in bold.</p>
</table-wrap-foot>
</table-wrap>
<p>To better investigate the classification process of each method, we visualized the performance of each method via the technique of CAM (<xref ref-type="bibr" rid="ref30">Selvaraju et al., 2017</xref>). The visual results of the comparison experiment are shown in <xref ref-type="fig" rid="fig7">Figure 7</xref>. From the visual result, we can intuitively observe that the proposed SPEMix is more adept at focusing on crucial information compared to other advanced data augmentation methods. Based on the comparison experiment results in <xref ref-type="table" rid="tab1">Table 1</xref> and <xref ref-type="fig" rid="fig7">Figure 7</xref>, we got the following conclusions:</p>
<list list-type="order">
<list-item>
<p>Our proposed SPEMix is less affected by the echocardiogram&#x2019;s background information than CutMix. Our SPEMix can improve by 2.91% with the encoder of Wideresnet and improve by 1.31% with the lightweight encoder compared to CutMix. To better understand the experimental results, we compared the visualization results of CutMix and SPEMix in <xref ref-type="fig" rid="fig7">Figure 7</xref>. These results illustrated that CutMix focuses more on the background information of the echocardiogram. This is mainly because cutmix performs mixup augmentation by randomly selecting regions of the image. This approach results in the appearance of augmentation data containing only background information, leading to the trained model not differentiating well between the foreground and background regions of a medical image. Hence, the classification performances of Cutmix are more affected by irrelevant information.</p>
</list-item>
<list-item>
<p>Our proposed SPEMix focused on more regions containing salient echocardiogram information than SaliencyMix. Our SPEMix improved by about 0.9% with the encoder of Wideresnet and about 0.41% with the encoder of RepViT compared to SaliencyMix. The saliency information helped our SPEMix and SaliencyMix focus on the vital information from the visual results in <xref ref-type="fig" rid="fig7">Figure 7</xref>. However, the SPEMix focused more regions on the salient information of the echocardiographic foreground region when performing view classification. The main reason is that our SPEMix can include more salient details in the foreground by seeking detailed salient information at the pixel level. However, SaliencyMix seeks salient areas at the image level, and this method cannot consider the detailed features.</p>
</list-item>
<list-item>
<p>Our proposed SPEMix sought vital information more easily from an echocardiogram than Automix. Our SPEMix can improve by 0.87% with the encoder of Wideresnet and by 0.37% with the encoder RepViT compared to AutoMix. These visualization results illustrated that SPEMix more easily focused on vital information. These results were mainly because our SPEMix generated the mixed mask with the assistance of dynamic attention. The mixed masks containing vital information helped the model easily seek important details. However, Automix generated the mixed mask only considering the similarity of the two images. The background information of the medical images was also similar between different views. This background information prevented the model from seeking vital information.</p>
</list-item>
<list-item>
<p>Furthermore, the experimental results also revealed the potential of the designed lightweight network for application in different methods. The designed lightweight network improved the performance of TMED2 compared to the traditional WideResnet encoder using various techniques for training. In CutMix, SaliencyMix, and AutoMix, the lightweight network enhanced by 1.67%, 0.56%, and 0.57%, respectively. Using our SPEMix method for training, the lightweight network still maintains better performance. The experimental results demonstrate the greater feasibility of lightweight networks in view classification.</p>
</list-item>
</list>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>The class activation mapping (CAM) for classifiers that are trained based on different mixing methods. The data of each view is chosen from the validation datasets randomly. The visualization results of regions focused on lightweight models trained by different methods. Our proposed SPEMix can better focus on the vital regions than other SOTA methods.</p>
</caption>
<graphic xlink:href="frai-07-1467218-g007.tif"/>
</fig>
<p>In summary, our SPEMix can achieve the best test classification accuracy on the WideResNet encoder and our lightweight encoder compared to other mixup methods. These experiment results demonstrate our SPEMix has advanced performance in view classification tasks.</p>
</sec>
<sec id="sec12">
<label>3.4</label>
<title>Comparison results with semi-supervised learning</title>
<p>In this section, we compared the performance of SPEMix with other competitive semi-supervised learning methods, which include FixMatch (<xref ref-type="bibr" rid="ref32">Sohn et al., 2020</xref>), Fix-a-step (<xref ref-type="bibr" rid="ref14">Huang et al., 2023</xref>), OpenMatch (<xref ref-type="bibr" rid="ref28">Saito et al., 2021</xref>), and InterLUDE (<xref ref-type="bibr" rid="ref15">Huang et al., 2024</xref>). We test different methods on TMED2, CAMUS, and Unity datasets to evaluate the accuracy and generalization of these methods. We trained the FixMatch and OpenMatch with the optimal parameters. Specifically, each method was trained with our lightweight encoder, the Adam optimizer, and 500 epochs. The learning rate was set to 0.001. We also trained Fix-a-step with the optimal parameters(Wideresnet as encoder and SGD optimizer) given in their paper to ensure that their method can achieve the most accurate classification results. For the InterLUDE and IntereLUDE+(IntereLUDE with Self-Adaptive Threshold and Self-Adaptive Fairness), we directly quote the results given in the paper(the code is not yet open source). The comparison results for accuracy and training time are presented in <xref ref-type="table" rid="tab2">Tables 2</xref>, <xref ref-type="table" rid="tab3">3</xref>, respectively.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>The performance of different semi-supervised methods on three datasets. All results are expressed as percentages (%).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Methods</th>
<th align="center" valign="top">TMED2 (<xref ref-type="bibr" rid="ref13">Huang et al., 2022</xref>)</th>
<th align="center" valign="top">CAMUS (<xref ref-type="bibr" rid="ref19">Leclerc et al., 2019</xref>)</th>
<th align="center" valign="top">Unity (<xref ref-type="bibr" rid="ref11">Howard et al., 2021</xref>)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">FixMatch (<xref ref-type="bibr" rid="ref32">Sohn et al., 2020</xref>)</td>
<td align="char" valign="top" char="&#x00B1;">95.49 &#x00B1; 0.07</td>
<td align="char" valign="top" char="&#x00B1;">81.39 &#x00B1; 0.72</td>
<td align="char" valign="top" char="&#x00B1;">91.41 &#x00B1; 1.28</td>
</tr>
<tr>
<td align="left" valign="top"><inline-formula>
<mml:math id="M104">
<mml:mi mathvariant="italic">Fix</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>a</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:msup>
<mml:mi>p</mml:mi>
<mml:mo>&#x2020;</mml:mo>
</mml:msup>
</mml:math>
</inline-formula> (<xref ref-type="bibr" rid="ref14">Huang et al., 2023</xref>)</td>
<td align="char" valign="top" char="&#x00B1;">95.64 &#x00B1; 0.16</td>
<td align="char" valign="top" char="&#x00B1;">83.20 &#x00B1; 0.74</td>
<td align="char" valign="top" char="&#x00B1;">93.72 &#x00B1; 0.65</td>
</tr>
<tr>
<td align="left" valign="top">OpenMatch (<xref ref-type="bibr" rid="ref28">Saito et al., 2021</xref>)</td>
<td align="char" valign="top" char="&#x00B1;">96.31 &#x00B1; 0.06</td>
<td align="char" valign="top" char="&#x00B1;">83.37 &#x00B1; 2.42</td>
<td align="char" valign="top" char="&#x00B1;">92.11 &#x00B1; 1.08</td>
</tr>
<tr>
<td align="left" valign="top">IntereLUDE&#x002A; (<xref ref-type="bibr" rid="ref15">Huang et al., 2024</xref>)</td>
<td align="char" valign="top" char="&#x00B1;">96.55 &#x00B1; 0.39</td>
<td align="char" valign="top" char="&#x00B1;">86.25 &#x00B1; 6.22</td>
<td align="char" valign="top" char="&#x00B1;"><bold>96.14 &#x00B1; 0.48</bold></td>
</tr>
<tr>
<td align="left" valign="top">IntereLUDE+&#x002A; (<xref ref-type="bibr" rid="ref15">Huang et al., 2024</xref>)</td>
<td align="char" valign="top" char="&#x00B1;">96.75 &#x00B1; 0.17</td>
<td align="char" valign="top" char="&#x00B1;">81.88 &#x00B1; 8.37</td>
<td align="char" valign="top" char="&#x00B1;">94.47 &#x00B1; 0.85</td>
</tr>
<tr>
<td align="left" valign="top">SPEMix(Ours)</td>
<td align="char" valign="top" char="&#x00B1;"><bold>97.28 &#x00B1; 0.11</bold></td>
<td align="char" valign="top" char="&#x00B1;"><bold>87.64 &#x00B1; 0.67</bold></td>
<td align="char" valign="top" char="&#x00B1;">94.71 &#x00B1; 0.24</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><sup>&#x2020;</sup>Means the method follows the parameters of the cited work with the encoder of Wideresnet and uses SGD optimizer. &#x002A;Means the results cited from the SOTA work. Other methods used our lightweight encoder and the Adam optimizer. The best performances of different datasets in bold.</p>
</table-wrap-foot>
</table-wrap>
<table-wrap position="float" id="tab3">
<label>Table 3</label>
<caption>
<p>The comparison results of training time between the traditional semi-supervised method and SPEMix.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Method</th>
<th align="center" valign="top">Total time</th>
<th align="center" valign="top">Average time</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">OpenMatch</td>
<td align="char" valign="top" char=".">4595.57&#x202F;s</td>
<td align="char" valign="top" char=".">15.32&#x202F;s</td>
</tr>
<tr>
<td align="left" valign="top">FixMatch</td>
<td align="char" valign="top" char=".">2746.83&#x202F;s</td>
<td align="char" valign="top" char=".">9.16&#x202F;s</td>
</tr>
<tr>
<td align="left" valign="top">SPEMix(Ours)</td>
<td align="char" valign="top" char=".">2714.18&#x202F;s</td>
<td align="char" valign="top" char=".">9.05&#x202F;s</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From the comparison results, we found that our SPEMix can achieve better accuracy on the three datasets. Compared to the FixMatch, the accuracy, and generalization of our SPEMix are improved by 1.79%, 6.25%, and 3.3% on TMED2, CAMUS, and Unity, respectively. The reason is that the proposed SP Block can leverage the out-of-distribution data to improve the accuracy. FixMatch focuses on the close-set problem and does not pay attention to the out-of-distribution echocardiograms so FixMatch has worse results and is not suitable for open-set tasks (<xref ref-type="bibr" rid="ref28">Saito et al., 2021</xref>). Meanwhile compared to OpenMatch, the accuracy of our SPEMix is improved by 0.97%, 4.27%, and 2.6% on TMED2, CAMUS, and Unity, respectively. The reason is that our SPEMix can also gain more improvement from the unlabeled datasets via the proposed mixed unlabeled consistency regularization. However, OpenMatch filtered out out-of-distribution data during the training process via its consistency regularization to process the open set data. In this way, OpenMatch neglected some vital information of out-of-distribution so that it has worse results. Compared to the advanced method for similar medical tasks, Fix-a-step, the SPEMix also improves the classification accuracy on different three datasets. Specifically, the accuracy and generalization of SPEMix compared to Fix-a-step improved by 1.64%, 4.44%, and 0.99% on TMED2, CAMUS and Unity, respectively. The reason for these results is that our SPEMix modeled the out-of-distribution data using superclass distribution to utilize all of the unlabeled data efficiently. However, Fix-a-step only used limited unlabeled data, which can improve the classification accuracy. In this way, Fix-a-step only got finite information from the limited out-of-distribution data. Simultaneously, our SPEMix also compares recent SOTA semi-supervised methods in view classification tasks, IntereLUDE and IntereLUDE+(IntereLUDE with Self-Adaptive Threshold and Self-Adaptive Fairness). Compared to the IntereLUDE, our SPEMix can improve by 0.73% and 1.39% on TMED2 and CAMUS. Compared to the IntereLUDE+, our SPEMix can improve by 0.53%, 5.76%, and 0.24% on TMED2, CAMUS, and Unity, respectively. SPEMix has less accuracy in the Unity dataset than IntereLUDE. However, IntereLUDE evaluated the accuracy on the widereset and did not explore a lightweight model. Therefore, our SPEMix still outperforms IntereLUDE overall.</p>
<p>These demonstrate that the accuracy and generalization of our proposed SPEMix outperform the SOTA methods in the view classification task. In summary, all of these results indicate the superior performance and generalization ability of the proposed SPEMix.</p>
</sec>
<sec id="sec13">
<label>3.5</label>
<title>Comparison results between different encoders</title>
<p>In this section, we explored the performance of our proposed lightweight encoder. We reported the number of parameters of the different encoders used in our comparison experiments of Section 3.3, i.e., the number of the parameters of the WideResNet-28-2 and our proposed lightweight encoder. We also reported the accuracy of the two encoders on the TMED2 test dataset after training via SPEMix. The comparison results were shown in <xref ref-type="table" rid="tab4">Table 4</xref>.</p>
<table-wrap position="float" id="tab4">
<label>Table 4</label>
<caption>
<p>The comparison results of different encoders in our comparison experiments. All accuracies are expressed as percentages (%).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">Encoder model</th>
<th align="center" valign="top">Parameters</th>
<th align="center" valign="top">Accuracy</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">WideResNet-28-2 (<xref ref-type="bibr" rid="ref37">Zagoruyko and Komodakis, 2016</xref>)</td>
<td align="char" valign="top" char=".">5.93&#x202F;M</td>
<td align="char" valign="top" char=".">97.24</td>
</tr>
<tr>
<td align="left" valign="top">Lightweight encoder(Ours)</td>
<td align="char" valign="top" char=".">0.70&#x202F;M</td>
<td align="char" valign="top" char=".">97.34</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The results in <xref ref-type="table" rid="tab4">Table 4</xref> demonstrated that the parameter number of our proposed encoder is reduced by 88.2% compared with the parameter number of the WideResNet-28-2. Additionally, the results also demonstrated that our proposed lightweight encoder can also maintain the classification performance compared with the WideResNet-28-2 when training via the proposed SPEMix. Otherwise, these results of <xref ref-type="table" rid="tab1">Table 1</xref> also denoted that, with different mixing methods, our proposed lightweight encoder was better than the WideResNet-28-2. These results demonstrated that our proposed SPEMix got better performance with fewer parameters. Hence, our proposed method had the potential for clinical application.</p>
</sec>
<sec id="sec14">
<label>3.6</label>
<title>Results of ablation experiments</title>
<p>The proposed SPEMix included two core components, DAMix and SP Block, to perform the mixed data augmentation and superclass pseudo-label generation, respectively. In order to explore the effect of each component in SPEMix, the lightweight network was regarded as the baseline in the ablation experiment, and we added DAMix Block and SP Block to the baseline step by step. The final results of the ablation experiments were in <xref ref-type="table" rid="tab5">Table 5</xref>.</p>
<table-wrap position="float" id="tab5">
<label>Table 5</label>
<caption>
<p>The results of the ablation experiment. All results are expressed as percentages (%).</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top">TMED2 (<xref ref-type="bibr" rid="ref13">Huang et al., 2022</xref>)</th>
<th align="center" valign="top">CAMUS (<xref ref-type="bibr" rid="ref19">Leclerc et al., 2019</xref>)</th>
<th align="center" valign="top">Unity (<xref ref-type="bibr" rid="ref11">Howard et al., 2021</xref>)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Baseline</td>
<td align="char" valign="top" char=".">94.2</td>
<td align="char" valign="top" char=".">74.93</td>
<td align="char" valign="top" char=".">85.07</td>
</tr>
<tr>
<td align="left" valign="top">Baseline+DAMix</td>
<td align="char" valign="top" char=".">96.02</td>
<td align="char" valign="top" char=".">81.56</td>
<td align="char" valign="top" char=".">91.11</td>
</tr>
<tr>
<td align="left" valign="top">Baseline+SP Block</td>
<td align="char" valign="top" char=".">96.21</td>
<td align="char" valign="top" char=".">82.56</td>
<td align="char" valign="top" char=".">93.36</td>
</tr>
<tr>
<td align="left" valign="top">SPEMix</td>
<td align="char" valign="top" char=".">97.34</td>
<td align="char" valign="top" char=".">87.11</td>
<td align="char" valign="top" char=".">94.42</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The baseline was the proposed lightweight encoder. We trained the baseline by using the TMED2 and reported the classification accuracy on different test datasets.</p>
</table-wrap-foot>
</table-wrap>
<p>Through the results of the ablation experiment, we can observe that the DAMix Block improved the baseline by 1.82%, 6.63%, and 6.04% on TMED2, CAMUS, and Unity. These results demonstrated the Efficient Mixup based on our proposed DAMix Block can improve accuracy. Furthermore, the SP Block improved the baseline by 2.01%, 0.73%, and 8.29% on TMED2, CAMUS, and Unity. The results denoted the advancement of utilizing the out-of-distribution data from the superclass probability perspective. The SPEMix can increase the baseline by 3.14%, 12.18%, and 9.35% on TMED2, CAMUS, and Unity, which illustrates the SPEMix can fuse the advantages of the DAMix Block and SP Block. These results demonstrated the effectiveness of the proposed SPEMix. The DAMix Block can generate corresponding mixed masks embedded with a mixing ratio for unlabelled data. The high-quality mixed data generated through the mixed masks can improve classification accuracy and generalization. SP Block can assign superclass pseudo-labels to unlabelled data through the perspective of superclass distribution to make use of the out-of-distribution data. This out-of-distribution information improved the classification accuracy and generilization. SP Block used an end-to-end approach to fuse two phases. Firstly, the unlabelled mixed data was generated using DAMix, while the information of generated mixed data was leveraged by assigning superclass pseudo-labels through SP Block. The high-quality mixed data generated by DAMix can also enrich the information of out-of-distribution that be leveraged by the SP Block.</p>
</sec>
</sec>
<sec sec-type="conclusions" id="sec15">
<label>4</label>
<title>Conclusion</title>
<p>In this work, we proposed a novel lightweight open-set semi-supervised learning method, SPEMix, to improve the accuracy and generalization of the echocardiogram view classification. The proposed DAMix Block generated the masks embedded with the mixing ratio containing vital information. These masks efficiently generated high-quality mixed data at the pixel level. Then, the proposed SP Block can generate the superclass pseudo-label from the superclass probability perspective to utilize the vital information from the unlabeled medical dataset. In this way, the proposed SP Block used the out-of-distribution data more effectively. Meanwhile, a novel loss function based on unlabeled consistent regularization was proposed to make the classification model better optimized from the supervision of the mixed unlabeled data and the super pseudo-label. Otherwise, we built a lightweight encoder based on RepViT to decrease the model parameters and improve the classification efficiency. Experiment results indicated that our proposed SPEMix achieved better performance and generalization than other semi-supervised learning methods. Our SPEMix had the potential for clinical application. Although the proposed SPEMix method shows encouraging results for echocardiogram view classification, there are still several areas that require further investigation. Future research could aim to adapt the method for more complex or multi-modal medical datasets, incorporating additional imaging modalities (such as CT or MRI) and patient metadata into the semi-supervised framework.</p>
</sec>
</body>
<back>
<sec sec-type="data-availability" id="sec16">
<title>Data availability statement</title>
<p>The Tufts Medical Echocardiogram Dataset 2 (TMED2) (<xref ref-type="bibr" rid="ref13">Huang et al., 2022</xref>) was used to train the proposed SPEMix. The CAMUS (<xref ref-type="bibr" rid="ref19">Leclerc et al., 2019</xref>) dataset and Unity (<xref ref-type="bibr" rid="ref11">Howard et al., 2021</xref>) dataset were used to evaluate the generalization.</p>
</sec>
<sec sec-type="author-contributions" id="sec17">
<title>Author contributions</title>
<p>SM: Conceptualization, Data curation, Methodology, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. YZ: Data curation, Methodology, Software, Writing &#x2013; review &#x0026; editing. DL: Methodology, Software, Writing &#x2013; review &#x0026; editing. YS: Formal analysis, Writing &#x2013; review &#x0026; editing. ZQ: Validation, Funding acquisition, Writing &#x2013; review &#x0026; editing. LW: Validation, Funding acquisition, Writing &#x2013; review &#x0026; editing. SD: Conceptualization, Project administration, Resources, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing.</p>
</sec>
<sec sec-type="funding-information" id="sec18">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This work was financially supported by the National Natural Science Foundation of China under Grant (no. 62202092), the Fundamental Research Funds for the Central Universities (no. 2572021BH03), Heilongjiang Provincial Key Research and Development Plan 2023ZX02C10, 2022ZX01A30, and GA23C007, Hunan Provincial Key Research and Development Plan 2023SK2060, Jiangsu Provincial Key Research and Development Plan BE2023081. All the above funders provide the data access support for this research.</p>
</sec>
<sec sec-type="COI-statement" id="sec19">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="sec20">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Avola</surname> <given-names>D.</given-names></name> <name><surname>Cannistraci</surname> <given-names>I.</given-names></name> <name><surname>Cascio</surname> <given-names>M.</given-names></name> <name><surname>Cinque</surname> <given-names>L.</given-names></name> <name><surname>Fagioli</surname> <given-names>A.</given-names></name> <name><surname>Foresti</surname> <given-names>G. L.</given-names></name> <etal/></person-group>. (<year>2024</year>). <article-title>MV-MS-FETE: multi-view multi-scale feature extractor and transformer encoder for stenosis recognition in echocardiograms</article-title>. <source>Comput. Methods Prog. Biomed.</source> <volume>245</volume>:<fpage>108037</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cmpb.2024.108037</pub-id>, PMID: <pub-id pub-id-type="pmid">38271793</pub-id></citation></ref>
<ref id="ref2"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Bai</surname> <given-names>W.</given-names></name> <name><surname>Oktay</surname> <given-names>O.</given-names></name> <name><surname>Sinclair</surname> <given-names>M.</given-names></name> <name><surname>Suzuki</surname> <given-names>H.</given-names></name> <name><surname>Rajchl</surname> <given-names>M.</given-names></name> <name><surname>Tarroni</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2017</year>). <italic>Semi-supervised learning for network-based cardiac MR image segmentation</italic>. Medical Image Computing and Computer-Assisted Intervention&#x2212; MICCAI 2017: 20th International Conference, Quebec City, QC, Canada, September 11&#x2013;13, 2017, Proceedings, Part II 20; Springer.</citation></ref>
<ref id="ref3"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Calderon-Ramirez</surname> <given-names>S.</given-names></name> <name><surname>Yang</surname> <given-names>S.</given-names></name> <name><surname>Elizondo</surname> <given-names>D.</given-names></name></person-group> (<year>2022</year>). <article-title>Semisupervised deep learning for image classification with distribution mismatch: a survey</article-title>. <source>IEEE Trans. Artif. Intell.</source> <volume>3</volume>, <fpage>1015</fpage>&#x2013;<lpage>1029</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TAI.2022.3196326</pub-id></citation></ref>
<ref id="ref4"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Chebli</surname> <given-names>A.</given-names></name> <name><surname>Djebbar</surname> <given-names>A.</given-names></name> <name><surname>Marouani</surname> <given-names>H. F.</given-names></name></person-group> (<year>2018</year>). <italic>Semi-supervised learning for medical application: A survey</italic>. 2018 international conference on applied smart systems (ICASS), IEEE.</citation></ref>
<ref id="ref5"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Degerli</surname> <given-names>A.</given-names></name> <name><surname>Kiranyaz</surname> <given-names>S.</given-names></name> <name><surname>Hamid</surname> <given-names>T.</given-names></name> <name><surname>Mazhar</surname> <given-names>R.</given-names></name> <name><surname>Gabbouj</surname> <given-names>M.</given-names></name></person-group> (<year>2024</year>). <article-title>Early myocardial infarction detection over multi-view echocardiography</article-title>. <source>Biomed. Signal Proc. Control</source> <volume>87</volume>:<fpage>105448</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.bspc.2023.105448</pub-id></citation></ref>
<ref id="ref6"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Ding</surname> <given-names>X.</given-names></name> <name><surname>Guo</surname> <given-names>Y.</given-names></name> <name><surname>Ding</surname> <given-names>G.</given-names></name> <name><surname>Han</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <italic>Acnet: Strengthening the kernel skeletons for powerful cnn via asymmetric convolution blocks</italic>. Proceedings of the IEEE/CVF international conference on computer vision.</citation></ref>
<ref id="ref7"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>W.</given-names></name> <name><surname>Loomes</surname> <given-names>M.</given-names></name> <name><surname>Wang</surname> <given-names>L.</given-names></name></person-group> (<year>2017</year>). <article-title>A fused deep learning architecture for viewpoint classification of echocardiography</article-title>. <source>Inf. Fusion</source> <volume>36</volume>, <fpage>103</fpage>&#x2013;<lpage>113</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.inffus.2016.11.007</pub-id></citation></ref>
<ref id="ref8"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hagberg</surname> <given-names>E.</given-names></name> <name><surname>Hagerman</surname> <given-names>D.</given-names></name> <name><surname>Johansson</surname> <given-names>R.</given-names></name> <name><surname>Hosseini</surname> <given-names>N.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name> <name><surname>Bj&#x00F6;rnsson</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Semi-supervised learning with natural language processing for right ventricle classification in echocardiography&#x2014;a scalable approach</article-title>. <source>Comput. Biol. Med.</source> <volume>143</volume>:<fpage>105282</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.compbiomed.2022.105282</pub-id>, PMID: <pub-id pub-id-type="pmid">35220074</pub-id></citation></ref>
<ref id="ref9"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>Y.</given-names></name> <name><surname>Duan</surname> <given-names>B.</given-names></name> <name><surname>Guan</surname> <given-names>R.</given-names></name> <name><surname>Yang</surname> <given-names>G.</given-names></name> <name><surname>Zhen</surname> <given-names>Z.</given-names></name></person-group> (<year>2024</year>). <article-title>LUFFD-YOLO: a lightweight model for UAV remote sensing Forest fire detection based on attention mechanism and multi-level feature fusion</article-title>. <source>Remote Sens.</source> <volume>16</volume>:<fpage>2177</fpage>. doi: <pub-id pub-id-type="doi">10.3390/rs16122177</pub-id></citation></ref>
<ref id="ref10"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>Y.</given-names></name> <name><surname>Guo</surname> <given-names>J.</given-names></name> <name><surname>Yang</surname> <given-names>H.</given-names></name> <name><surname>Guan</surname> <given-names>R.</given-names></name> <name><surname>Zhang</surname> <given-names>T.</given-names></name></person-group> (<year>2024</year>). <article-title>SSMA-YOLO: a lightweight YOLO model with enhanced feature extraction and fusion capabilities for drone-aerial ship image detection</article-title>. <source>Drones</source> <volume>8</volume>:<fpage>145</fpage>. doi: <pub-id pub-id-type="doi">10.3390/drones8040145</pub-id></citation></ref>
<ref id="ref11"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Howard</surname> <given-names>J. P.</given-names></name> <name><surname>Stowell</surname> <given-names>C. C.</given-names></name> <name><surname>Cole</surname> <given-names>G. D.</given-names></name> <name><surname>Ananthan</surname> <given-names>K.</given-names></name> <name><surname>Demetrescu</surname> <given-names>C. D.</given-names></name> <name><surname>Pearce</surname> <given-names>K.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Automated left ventricular dimension assessment using artificial intelligence developed and validated by a UK-wide collaborative</article-title>. <source>Circ. Cardiovasc. Imaging</source> <volume>14</volume>:<fpage>e011951</fpage>. doi: <pub-id pub-id-type="doi">10.1161/CIRCIMAGING.120.011951</pub-id>, PMID: <pub-id pub-id-type="pmid">33998247</pub-id></citation></ref>
<ref id="ref12"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>Z.</given-names></name> <name><surname>Long</surname> <given-names>G.</given-names></name> <name><surname>Wessler</surname> <given-names>B.</given-names></name> <name><surname>Hughes</surname> <given-names>M. C.</given-names></name></person-group> <italic>A new semi-supervised learning benchmark for classifying view and diagnosing aortic stenosis from echocardiograms</italic>. Machine Learning for Healthcare Conference, PMLR. (<year>2021</year>).</citation></ref>
<ref id="ref13"><citation citation-type="other"><person-group person-group-type="editor"><name><surname>Huang</surname> <given-names>Z.</given-names></name> <name><surname>Long</surname> <given-names>G.</given-names></name> <name><surname>Wessler</surname> <given-names>B.</given-names></name> <name><surname>Hughes</surname> <given-names>M. C.</given-names></name></person-group> (<year>2022</year>). <italic>TMED 2: A dataset for semi-supervised classification of echocardiograms</italic>. DataPerf: Benchmarking Data for Data-Centric AI Workshop.</citation></ref>
<ref id="ref14"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>Z.</given-names></name> <name><surname>Sidhom</surname> <given-names>M. J.</given-names></name> <name><surname>Wessler</surname> <given-names>B.</given-names></name> <name><surname>Hughes</surname> <given-names>M. C.</given-names></name></person-group> (<year>2023</year>). <italic>Fix-A-step: semi-supervised learning from Uncurated unlabeled data</italic>. International Conference on Artificial Intelligence and Statistics, PMLR.</citation></ref>
<ref id="ref15"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>Z.</given-names></name> <name><surname>Yu</surname> <given-names>X.</given-names></name> <name><surname>Zhu</surname> <given-names>D.</given-names></name> <name><surname>Hughes</surname> <given-names>M. C.</given-names></name></person-group> (<year>2024</year>). <article-title>InterLUDE: interactions between labeled and unlabeled data to enhance semi-supervised learning</article-title>. <source>arXiv</source> <volume>2024</volume>:<fpage>240310658</fpage>. doi: <pub-id pub-id-type="doi">10.48550/arXiv.2403.10658</pub-id></citation></ref>
<ref id="ref16"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kusunose</surname> <given-names>K.</given-names></name> <name><surname>Haga</surname> <given-names>A.</given-names></name> <name><surname>Inoue</surname> <given-names>M.</given-names></name> <name><surname>Fukuda</surname> <given-names>D.</given-names></name> <name><surname>Yamada</surname> <given-names>H.</given-names></name> <name><surname>Sata</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Clinically feasible and accurate view classification of echocardiographic images using deep learning</article-title>. <source>Biomol. Ther.</source> <volume>10</volume>:<fpage>665</fpage>. doi: <pub-id pub-id-type="doi">10.3390/biom10050665</pub-id>, PMID: <pub-id pub-id-type="pmid">32344829</pub-id></citation></ref>
<ref id="ref17"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kwon</surname> <given-names>H.</given-names></name></person-group> (<year>2023</year>). <article-title>Adversarial image perturbations with distortions weighted by color on deep neural networks</article-title>. <source>Multimed. Tools Appl.</source> <volume>82</volume>, <fpage>13779</fpage>&#x2013;<lpage>13795</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11042-022-12941-w</pub-id></citation></ref>
<ref id="ref18"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kwon</surname> <given-names>H.</given-names></name> <name><surname>Lee</surname> <given-names>S.</given-names></name></person-group> (<year>2023</year>). <article-title>Detecting textual adversarial examples through text modification on text classification systems</article-title>. <source>Appl. Intell.</source> <volume>53</volume>, <fpage>19161</fpage>&#x2013;<lpage>19185</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s10489-022-03313-w</pub-id></citation></ref>
<ref id="ref19"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leclerc</surname> <given-names>S.</given-names></name> <name><surname>Smistad</surname> <given-names>E.</given-names></name> <name><surname>Pedrosa</surname> <given-names>J.</given-names></name> <name><surname>&#x00D8;stvik</surname> <given-names>A.</given-names></name> <name><surname>Cervenansky</surname> <given-names>F.</given-names></name> <name><surname>Espinosa</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Deep learning for segmentation using an open large-scale dataset in 2D echocardiography</article-title>. <source>IEEE Trans. Med. Imaging</source> <volume>38</volume>, <fpage>2198</fpage>&#x2013;<lpage>2210</lpage>. doi: <pub-id pub-id-type="doi">10.1109/TMI.2019.2900516</pub-id>, PMID: <pub-id pub-id-type="pmid">30802851</pub-id></citation></ref>
<ref id="ref20"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>J.</given-names></name> <name><surname>Kim</surname> <given-names>T.</given-names></name> <name><surname>Bang</surname> <given-names>S.</given-names></name> <name><surname>Oh</surname> <given-names>S.</given-names></name> <name><surname>Kwon</surname> <given-names>H.</given-names></name></person-group> (<year>2024</year>). <article-title>Evasion attacks on deep learning-based helicopter recognition systems</article-title>. <source>J Sens</source> <volume>2024</volume>, <fpage>1</fpage>&#x2013;<lpage>9</lpage>. doi: <pub-id pub-id-type="doi">10.1155/2024/1124598</pub-id></citation></ref>
<ref id="ref21"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Qi</surname> <given-names>L.</given-names></name> <name><surname>Shi</surname> <given-names>Y.</given-names></name> <name><surname>Gao</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <italic>IOMatch: Simplifying open-set semi-supervised learning with joint inliers and outliers utilization</italic>. Proceedings of the IEEE/CVF International Conference on Computer Vision.</citation></ref>
<ref id="ref22"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Wu</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>Z.</given-names></name> <name><surname>Wu</surname> <given-names>L.</given-names></name> <etal/></person-group>. (<year>2022</year>). <italic>Automix: Unveiling the power of mixup for stronger classifiers</italic>. European Conference on Computer Vision, Springer.</citation></ref>
<ref id="ref23"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Loshchilov</surname> <given-names>I.</given-names></name> <name><surname>Hutter</surname> <given-names>F.</given-names></name></person-group> (<year>2022</year>). <italic>SGDR: Stochastic gradient descent with warm restarts</italic>. International conference on learning representations.</citation></ref>
<ref id="ref24"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Madani</surname> <given-names>A.</given-names></name> <name><surname>Arnaout</surname> <given-names>R.</given-names></name> <name><surname>Mofrad</surname> <given-names>M.</given-names></name> <name><surname>Arnaout</surname> <given-names>R.</given-names></name></person-group> (<year>2018</year>). <article-title>Fast and accurate view classification of echocardiograms using deep learning</article-title>. <source>NPJ Digit. Med.</source> <volume>1</volume>:<fpage>6</fpage>. doi: <pub-id pub-id-type="doi">10.1038/s41746-017-0013-1</pub-id>, PMID: <pub-id pub-id-type="pmid">30828647</pub-id></citation></ref>
<ref id="ref25"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Madani</surname> <given-names>A.</given-names></name> <name><surname>Ong</surname> <given-names>J. R.</given-names></name> <name><surname>Tibrewal</surname> <given-names>A.</given-names></name> <name><surname>Mofrad</surname> <given-names>M. R.</given-names></name></person-group> (<year>2018</year>). <article-title>Deep echocardiography: data-efficient supervised and semi-supervised deep learning towards automated diagnosis of cardiac disease</article-title>. <source>NPJ Digit. Med.</source> <volume>1</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41746-018-0065-x</pub-id></citation></ref>
<ref id="ref26"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Oliver</surname> <given-names>A.</given-names></name> <name><surname>Odena</surname> <given-names>A.</given-names></name> <name><surname>Raffel</surname> <given-names>C. A.</given-names></name> <name><surname>Cubuk</surname> <given-names>E. D.</given-names></name> <name><surname>Goodfellow</surname> <given-names>I.</given-names></name></person-group> (<year>2018</year>). <article-title>Realistic evaluation of deep semi-supervised learning algorithms</article-title>. <source>Adv. Neural Inf. Proces. Syst.</source> <volume>31</volume>:<fpage>11</fpage>.</citation></ref>
<ref id="ref27"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Polyak</surname> <given-names>B. T.</given-names></name> <name><surname>Juditsky</surname> <given-names>A. B.</given-names></name></person-group> (<year>1992</year>). <article-title>Acceleration of stochastic approximation by averaging</article-title>. <source>SIAM J. Control. Optim.</source> <volume>30</volume>, <fpage>838</fpage>&#x2013;<lpage>855</lpage>. doi: <pub-id pub-id-type="doi">10.1137/0330046</pub-id></citation></ref>
<ref id="ref28"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Saito</surname> <given-names>K.</given-names></name> <name><surname>Kim</surname> <given-names>D.</given-names></name> <name><surname>Saenko</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>Openmatch: open-set semi-supervised learning with open-set consistency regularization</article-title>. <source>Adv. Neural Inf. Proces. Syst.</source> <volume>34</volume>, <fpage>25956</fpage>&#x2013;<lpage>25967</lpage>.</citation></ref>
<ref id="ref29"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Saito</surname> <given-names>K.</given-names></name> <name><surname>Saenko</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <italic>Ovanet: One-vs-all network for universal domain adaptation</italic>. Proceedings of the ieee/cvf international conference on computer vision.</citation></ref>
<ref id="ref30"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Selvaraju</surname> <given-names>R. R.</given-names></name> <name><surname>Cogswell</surname> <given-names>M.</given-names></name> <name><surname>Das</surname> <given-names>A.</given-names></name> <name><surname>Vedantam</surname> <given-names>R.</given-names></name> <name><surname>Parikh</surname> <given-names>D.</given-names></name> <name><surname>Batra</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <italic>Grad-cam: Visual explanations from deep networks via gradient-based localization</italic>. Proceedings of the IEEE international conference on computer vision.</citation></ref>
<ref id="ref31"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shorten</surname> <given-names>C.</given-names></name> <name><surname>Khoshgoftaar</surname> <given-names>T. M.</given-names></name></person-group> (<year>2019</year>). <article-title>A survey on image data augmentation for deep learning</article-title>. <source>J. Big Data</source> <volume>6</volume>, <fpage>1</fpage>&#x2013;<lpage>48</lpage>. doi: <pub-id pub-id-type="doi">10.1186/s40537-019-0197-0</pub-id></citation></ref>
<ref id="ref32"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sohn</surname> <given-names>K.</given-names></name> <name><surname>Berthelot</surname> <given-names>D.</given-names></name> <name><surname>Carlini</surname> <given-names>N.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Raffel</surname> <given-names>C. A.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Fixmatch: simplifying semi-supervised learning with consistency and confidence</article-title>. <source>Adv. Neural Inf. Proces. Syst.</source> <volume>33</volume>, <fpage>596</fpage>&#x2013;<lpage>608</lpage>.</citation></ref>
<ref id="ref33"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Uddin</surname> <given-names>A. S.</given-names></name> <name><surname>Monira</surname> <given-names>M. S.</given-names></name> <name><surname>Shin</surname> <given-names>W.</given-names></name> <name><surname>Chung</surname> <given-names>T.</given-names></name> <name><surname>Bae</surname> <given-names>S. H.</given-names></name></person-group> (<year>2006</year>). <italic>SaliencyMix: a saliency guided data augmentation strategy for better regularization</italic>. International Conference on Learning Representations.</citation></ref>
<ref id="ref34"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>A.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Lin</surname> <given-names>Z.</given-names></name> <name><surname>Han</surname> <given-names>J.</given-names></name></person-group>, <person-group person-group-type="editor"><name><surname>Ding</surname> <given-names>G.</given-names></name></person-group> (<year>2024</year>). <italic>Repvit: Revisiting mobile cnn from vit perspective</italic>. Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition.</citation></ref>
<ref id="ref35"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Xiaojin</surname> <given-names>Z.</given-names></name></person-group> (<year>2008</year>). <italic>Semi-supervised learning literature survey</italic>. Computer Sciences TR, No. 1530.</citation></ref>
<ref id="ref36"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Yun</surname> <given-names>S.</given-names></name> <name><surname>Han</surname> <given-names>D.</given-names></name> <name><surname>Oh</surname> <given-names>S. J.</given-names></name> <name><surname>Chun</surname> <given-names>S.</given-names></name> <name><surname>Choe</surname> <given-names>J.</given-names></name> <name><surname>Yoo</surname> <given-names>Y.</given-names></name></person-group> (<year>2019</year>). <italic>Cutmix: Regularization strategy to train strong classifiers with localizable features</italic>. Proceedings of the IEEE/CVF international conference on computer vision.</citation></ref>
<ref id="ref37"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zagoruyko</surname> <given-names>S.</given-names></name> <name><surname>Komodakis</surname> <given-names>N.</given-names></name></person-group> (<year>2016</year>). <italic>Wide residual networks</italic>. British Machine Vision Conference 2016; British Machine Vision Association.</citation></ref>
<ref id="ref38"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Cisse</surname> <given-names>M.</given-names></name> <name><surname>Dauphin</surname> <given-names>Y. N.</given-names></name> <name><surname>Lopez-Paz</surname> <given-names>D.</given-names></name></person-group> (<year>2018</year>). <italic>Mixup: Beyond empirical risk minimization</italic>. International conference on learning representations.</citation></ref>
<ref id="ref39"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>X.</given-names></name> <name><surname>Krishnateja</surname> <given-names>K.</given-names></name> <name><surname>Iyer</surname> <given-names>R.</given-names></name> <name><surname>Chen</surname> <given-names>F.</given-names></name></person-group> (<year>2022</year>). <italic>How out-of-distribution data hurts semi-supervised learning</italic>. 2022 IEEE International Conference on Data Mining (ICDM), IEEE.</citation></ref>
<ref id="ref40"><citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Ma</surname> <given-names>J.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>Y.</given-names></name> <name><surname>Zhu</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2022</year>). <article-title>Automatic view classification of contrast and non-contrast echocardiography</article-title>. <source>Front. Cardiovasc. Med.</source> <volume>9</volume>:<fpage>989091</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fcvm.2022.989091</pub-id>, PMID: <pub-id pub-id-type="pmid">36186996</pub-id></citation></ref>
<ref id="ref41"><citation citation-type="other"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>L.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Ke</surname> <given-names>Z.</given-names></name> <name><surname>Zhang</surname> <given-names>W.</given-names></name> <name><surname>Lau</surname> <given-names>R. W.</given-names></name></person-group> (<year>2023</year>). <italic>Biformer: Vision transformer with bi-level routing attention</italic>. Proceedings of the IEEE/CVF conference on computer vision and pattern recognition.</citation></ref>
</ref-list>
</back>
</article>