<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mar. Sci.</journal-id>
<journal-title>Frontiers in Marine Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mar. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-7745</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmars.2024.1373755</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Marine Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Semi-supervised learning advances species recognition for aquatic biodiversity monitoring</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Ma</surname>
<given-names>Dongliang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2626428"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wei</surname>
<given-names>Jine</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhu</surname>
<given-names>Likai</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2323440"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Zhao</surname>
<given-names>Fang</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Wu</surname>
<given-names>Hao</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chen</surname>
<given-names>Xi</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/919735"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Li</surname>
<given-names>Ye</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Liu</surname>
<given-names>Min</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/464107"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Key Laboratory of Geographic Information Science, Ministry of Education, School of Geographic Sciences, East China Normal University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>State Key Laboratory of Estuarine and Coastal Research, East China Normal University</institution>, <addr-line>Shanghai</addr-line>, <country>China</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>School of Artificial Intelligence and Computer Science, Jiangnan University</institution>, <addr-line>Wuxi</addr-line>, <country>China</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Haiyong Zheng, Ocean University of China, China</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Zhuhua Hu, Hainan University, China</p>
<p>Deepayan Bhowmik, Newcastle University, United Kingdom</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Ye Li, <email xlink:href="mailto:yli@geo.ecnu.edu.cn">yli@geo.ecnu.edu.cn</email>; Min Liu, <email xlink:href="mailto:mliu@geo.ecnu.edu.cn">mliu@geo.ecnu.edu.cn</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>28</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1373755</elocation-id>
<history>
<date date-type="received">
<day>20</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>02</day>
<month>05</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Ma, Wei, Zhu, Zhao, Wu, Chen, Li and Liu</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Ma, Wei, Zhu, Zhao, Wu, Chen, Li and Liu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Aquatic biodiversity monitoring relies on species recognition from images. While deep learning (DL) streamlines the recognition process, the performance of these method is closely linked to the large-scale labeled datasets, necessitating manual processing with expert knowledge and consume substantial time, labor, and financial resources. Semi-supervised learning (SSL) offers a promising avenue to improve the performance of DL models by utilizing the extensive unlabeled samples. However, the complex collection environments and the long-tailed class imbalance of aquatic species make SSL difficult to implement effectively. To address these challenges in aquatic species recognition within the SSL scheme, we propose a Wavelet Fusion Network and the Consistency Equilibrium Loss function. The former mitigates the influence of data collection environment by fusing image information at different frequencies decomposed through wavelet transform. The latter improves the SSL scheme by refining the consistency loss function and adaptively adjusting the margin for each class. Extensive experiments are conducted on the large-scale FishNet dataset. As expected, our method improves the existing SSL scheme by up to 9.34% in overall classification accuracy. With the accumulation of image data, the improved SSL method with limited labeled data, shows the potential to advance species recognition for aquatic biodiversity monitoring and conservation.</p>
</abstract>
<kwd-group>
<kwd>deep learning</kwd>
<kwd>semi-supervised learning</kwd>
<kwd>aquatic species recognition</kwd>
<kwd>wavelet transform</kwd>
<kwd>consistency loss</kwd>
</kwd-group>
<counts>
<fig-count count="5"/>
<table-count count="7"/>
<equation-count count="10"/>
<ref-count count="63"/>
<page-count count="13"/>
<word-count count="7559"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Ocean Observation</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Aquatic biodiversity plays a crucial role in maintaining the structural integrity, stability, and overall health of ecosystems (<xref ref-type="bibr" rid="B44">Sala et&#xa0;al., 2021</xref>). However, anthropogenic pressures from human activities have progressively intensified in recent decades, posing gradual challenges to the preservation of aquatic biodiversity (<xref ref-type="bibr" rid="B53">Visbeck, 2018</xref>; <xref ref-type="bibr" rid="B16">Irfan and Alatawi, 2019</xref>). A critical step in conserving aquatic biodiversity is monitoring the information regarding the abundance and distribution of aquatic animals, which relies heavily on extensive collections of underwater images and videos. Deep learning (DL) techniques have recently demonstrated significant progress in several computer vision tasks (<xref ref-type="bibr" rid="B24">LeCun et&#xa0;al., 2015</xref>), and offer a promising solution to automatic and effective species recognition from images (<xref ref-type="bibr" rid="B43">Rubbens et&#xa0;al., 2023</xref>). Due to the profound influence of dataset size and diversity on the accuracy of DL methods, many previous efforts have focused on building extensive and publicly available labeled image datasets specifically for aquatic species recognition (<xref ref-type="bibr" rid="B63">Zhuang et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B18">Katija et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B20">Khan et&#xa0;al., 2023</xref>). Unfortunately, the intricate taxonomy of species typically demands a high level of expertise in the aquatic domain, meanwhile the annotation process proves to be tedious and time-consuming (<xref ref-type="bibr" rid="B28">Li et&#xa0;al., 2023</xref>).</p>
<p>It is estimated that more than 300,000 hours of underwater video footage have been collected worldwide so far, with only less than 15% of the data annotated by biological and ecological experts (<xref ref-type="bibr" rid="B1">Bell et&#xa0;al., 2023</xref>). As the pace of data collection accelerates annually, the substantial backlog exacerbates. Several strategies, including transfer learning (<xref ref-type="bibr" rid="B40">Qiu et&#xa0;al., 2018</xref>), data augmentation (<xref ref-type="bibr" rid="B45">Saleh et&#xa0;al., 2020</xref>), weakly supervised learning (<xref ref-type="bibr" rid="B23">Laradji et&#xa0;al., 2021</xref>), and active learning (<xref ref-type="bibr" rid="B37">Moller et&#xa0;al., 2017</xref>), have been made to tackle this problem. For example, transfer learning necessitates fine-tuning newly labeled aquatic species datasets to maximize accuracy. Weakly-supervised learning, on the other hand, relies on a limited form of supervision, where the labels may be noisy, incomplete, or imprecise. Nonetheless, these studies still require access to large-scale labeled training sets. The significance of diversity and comprehensiveness in the training dataset undoubtedly plays a pivotal role in achieving high recognition accuracy during real-world model deployment. Given the existence of unlabeled data, the marine community has emphasized the need for a powerful approach to training DL methods on vast amounts of data without annotated labels. In contrast, semi-supervised learning (SSL) can handle scenarios with both labeled and unlabeled data, providing more flexibility and potentially better performance when limited labeled data is available (<xref ref-type="bibr" rid="B57">Yang et&#xa0;al., 2022</xref>). To date, although numerous studies explore SSL to address the high cost of annotated labels in aquatic domain (<xref ref-type="bibr" rid="B7">Choi et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B4">Cai et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B17">Jahanbakht et&#xa0;al., 2023</xref>), its application in aquatic environments for species recognition remains scarce.</p>
<p>Two major challenges conspire to hinder the use of SSL scheme for aquatic species recognition. The first challenge stems from the unique characteristics of collected environments, including diverse lighting, variable water turbidity, and complex visual backgrounds that can obscure visual information (<xref ref-type="bibr" rid="B11">Ditria et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B47">Saleh et&#xa0;al., 2022</xref>). Furthermore, the movement of objects in an uncontrolled environment can introduce distortion, deformation, occlusion, and overlapping (<xref ref-type="bibr" rid="B28">Li et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B33">Ma et&#xa0;al., 2023</xref>). These factors increase complexities and hinder the ability of DL models to employ effectively from labeled to unlabeled data. The need for robust feature extraction methods tailored to the above challenges becomes paramount to ensure the practical applicability of the SSL scheme. The second challenge arises from the long-tailed class imbalance of aquatic species in collected images (<xref ref-type="bibr" rid="B43">Rubbens et&#xa0;al., 2023</xref>). As shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1A</bold>
</xref>, a limited subset of species are characterized by a substantial number of samples (referred to as head classes), while others are linked to only a few samples (referred to as tail classes). The limited sample information of tail classes poses a significant hurdle for SSL scheme, as there is a risk that the model being biased toward head classes due to the abundance of samples (<xref ref-type="bibr" rid="B60">Zhang et&#xa0;al., 2023</xref>).</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>
<bold>(A)</bold> The label distribution of a long-tailed aquatic species dataset (e.g., the FishNet dataset (<xref ref-type="bibr" rid="B20">Khan et&#xa0;al., 2023</xref>) with more than 450 classes). <bold>(B)</bold> Statistics of mean classification score for each class on the FishNet dataset. The x-axis depicts the index corresponding to the corresponding class.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1373755-g001.tif"/>
</fig>
<p>In this work, we propose a novel SSL scheme for aquatic species recognition, which is based on the existing SSL algorithm, FixMatch (<xref ref-type="bibr" rid="B48">Sohn et&#xa0;al., 2020</xref>). Specifically, to mitigate the complexities inherent in heterogeneous collected environments, we propose a robust wavelet fusion network (WFN) equipped with wavelet transform. The proposed network comprises two frequency-aware streams, one is dedicated to capturing subtle image details by focusing on high-frequency (HF) information, while the other aims to extract high-level semantics from low-frequency (LF) information. These streams are subsequently integrated through a FusionBlock, which facilitates attentive interactions between the LF and HF streams. Furthermore, for the problem of long-tailed nature when using unlabeled data, we design a new Consistency Equilibrium Loss (CEL) that refines the pseudo-labels and adaptively adjusts the margin for each aquatic species class. We find that replacing the unsupervised loss with CEL could ensure that the SSL algorithm achieves relative classification equilibrium, even if the collected data distribution is biased toward the head classes. Extensive experiments demonstrate the proposed method attains superior results on a large-scale aquatic species recognition dataset. In addition, the WFN and CEL are assessed to highlight their advantages over current common practices.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Related work</title>
<sec id="s2_1">
<label>2.1</label>
<title>Aquatic species recognition with deep learning</title>
<p>In recent years, DL-based aquatic species recognition has emerged as a promising tool for assisting marine scientists and ecologists in better understanding and managing marine environments. Accurate species recognition serves as the cornerstone of aquatic biodiversity research, playing a crucial role in estimating species size and quantity. A seminal contribution in this field is the development of the filtering deep convolutional network (FDCNet) (<xref ref-type="bibr" rid="B32">Lu et&#xa0;al., 2018</xref>), which effectively classifies deep-sea objects such as sea urchins, crabs, sharks, and shrimps. Due to the complexity and dynamics of the marine environment, DL methods encounter challenges in recognizing interesting objects based on visual characteristics. To overcome this issue, the literature (<xref ref-type="bibr" rid="B19">Kaur and Vijay, 2023</xref>) proposes an invariant feature-based species classification method for distinguishing octopus and crabs. Similarly, the study (<xref ref-type="bibr" rid="B30">Liu et&#xa0;al., 2023</xref>) introduces an improved fish recognition network along with a novel loss function, FishFace, designs to focus more attention on fish details. More recently, automated plankton recognizing method based on DL has been developed for continuous monitoring of living plankton abundance in aquatic environments (<xref ref-type="bibr" rid="B6">Chen et&#xa0;al., 2023</xref>). A comprehensive review (<xref ref-type="bibr" rid="B28">Li et&#xa0;al., 2023</xref>) is recommended for researchers to seek an in-depth understanding of DL-based aquatic species recognition methods. However, most existing methods are constrained by their reliance on a relatively small portion of labeled data, posing a challenge to their practical application in real-world scenarios (<xref ref-type="bibr" rid="B20">Khan et&#xa0;al., 2023</xref>). Therefore, there is an urgent need to develop a new paradigm capable of effectively utilizing extensive unlabeled data with a small amount of labeled data to accurately identify a broader range of aquatic species, thereby supporting aquatic biodiversity conservation efforts.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Semi-supervised learning</title>
<p>SSL methods have garnered significant attention from both industry and academia for use unlabeled data during the training process, particularly when the amount of labeled data is scarce. Recent SSL research has generally been categorized into two main groups. The first category of consistency regularization methods imposes a classification invariance loss on unlabeled data following perturbation (<xref ref-type="bibr" rid="B36">Miyato et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B55">Xie et&#xa0;al., 2020a</xref>). In the second category, pseudo-labeling extends model training data beyond labeled samples to contain additional unlabeled data, augmented with credible pseudo-labels (<xref ref-type="bibr" rid="B3">Berthelot et&#xa0;al., 2019b</xref>; <xref ref-type="bibr" rid="B56">Xie et&#xa0;al., 2020b</xref>). Techniques like FixMatch (<xref ref-type="bibr" rid="B48">Sohn et&#xa0;al., 2020</xref>) and RemixMatch (<xref ref-type="bibr" rid="B2">Berthelot et&#xa0;al., 2019a</xref>) combine pseudo-labeling with consistency regularization, yielding superior performance compared to many other SSL algorithms in image recognition tasks. Furthermore, several studies have been conducted experiments on long-tailed SSL. For example, DARP (<xref ref-type="bibr" rid="B21">Kim et&#xa0;al., 2020</xref>) proposes eliminating biased pseudo-labels through distribution alignment, which refines the pseudo-labels based on the labeled data distribution. Additionally, an auxiliary balanced classifier learned by down-sampling the head class is used to enhance generalization capabilities (<xref ref-type="bibr" rid="B25">Lee et&#xa0;al., 2021</xref>). The above designs largely promote the overall performance of long-tailed semi-supervised methods, but the performance of natural long-tailed SSL problems in aquatic species recognition is still unsatisfactory, and no research has been found that effectively addresses this issue.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Wavelet-based deep learning</title>
<p>The integration of wavelet transform with deep neural networks (DNNs) has gained traction due to its robust frequency and spatial representation capabilities. Common strategies involve utilizing wavelet transform as either a pre-processing or post-processing step (<xref ref-type="bibr" rid="B15">Huang et&#xa0;al., 2017</xref>; <xref ref-type="bibr" rid="B59">Yin and Xu, 2021</xref>), as well as substituting specific layers in DNNs (<xref ref-type="bibr" rid="B27">Li et&#xa0;al., 2021</xref>). Previous research has also explored the application of the dual-tree complex wavelet transform to extract robust features from Synthetic Aperture Radar images (<xref ref-type="bibr" rid="B12">Duan et&#xa0;al., 2017</xref>). More recently, Wave-ViT (<xref ref-type="bibr" rid="B58">Yao et&#xa0;al., 2022</xref>) uses the wavelet transform to down-sample keys/values in a Transformer (<xref ref-type="bibr" rid="B51">Vaswani et&#xa0;al., 2017</xref>). The Multi-level Wavelet CNN (<xref ref-type="bibr" rid="B31">Liu et&#xa0;al., 2018</xref>) integrates wavelet package transform into the DNN to concatenate the LF and HF components and process them in a unified manner, despite the notable disparity between these components. In contrast, we employ wavelet transform as an effective approach to tackle image complexity. Further, none of these studies has attempted to design a fusion block specially tailored for the wavelet transform paradigm to obtain attentive feature representation.</p>
</sec>
<sec id="s2_4">
<label>2.4</label>
<title>Loss function for long-tailed learning</title>
<p>Re-weighting and Re-margining loss functions serve as key components in tackling long-tailed class imbalanced challenges (<xref ref-type="bibr" rid="B60">Zhang et&#xa0;al., 2023</xref>). These methods are primarily implemented by adjusting margins or loss weights based on the distribution of training data. For instance, seminal works (<xref ref-type="bibr" rid="B8">Cui et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B42">Ren et&#xa0;al., 2020</xref>) reweight the loss functions according to the sampling frequency of each class. Recent literature (<xref ref-type="bibr" rid="B22">Lai et&#xa0;al., 2022</xref>) enhances the robustness of SSL to long-tailed class imbalanced problems by designing weights in the unsupervised loss based on estimating the learning difficulty of each class. In contrast, several studies (<xref ref-type="bibr" rid="B5">Cao et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B35">Menon et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B49">Tan et&#xa0;al., 2020</xref>) have attempted to adjust the loss margins of each class. The Label-Distribution-Aware-Margin (<xref ref-type="bibr" rid="B5">Cao et&#xa0;al., 2019</xref>) approach motivates tail classes to have larger margins based on label frequencies. Additionally, the study (<xref ref-type="bibr" rid="B13">Feng et&#xa0;al., 2021</xref>) replaces the margin term with mean classification score for long-tailed object detection. While our CEL function is inspired by the above pioneer studies, it differs significantly in two aspects. Firstly, to the best of our knowledge, the CEL function is the first to utilize the mean classification score to extend the existing consistency loss in SSL. Secondly, our key idea involves refining pseudo-labels via the mean classification score to match the true data distribution. With the proposed CEL function, our approach demonstrates superior performance in aquatic species recognition based on the SSL scheme.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Method</title>
<p>In this section, we first revisit the formulation of the SSL scheme in Section 3.1. After that, we illustrate the process of generating LF and HF entities using wavelet transform in Section 3.2, and provide detailed insights into our FusionBlock in Section 3.3. Lastly, along with the SSL scheme, we introduce the CEL function for unlabeled samples in Section 3.4. An overview of the framework is shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref>.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Illustration of the overall framework for aquatic species recognition. The proposed WFN (Wavelet Fusion Network) and CEL (Consistency Equilibrium Loss) are added into the exiting SSL scheme FixMatch (<xref ref-type="bibr" rid="B20">Khan et&#xa0;al., 2023</xref>).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1373755-g002.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Semi-supervised learning setup</title>
<p>The basic technique utilized in FixMatch (<xref ref-type="bibr" rid="B48">Sohn et&#xa0;al., 2020</xref>) revolves around pseudo-labeling and consistency-regularization, where unlabeled samples with high confidence are selected as training samples. Suppose we have a labeled dataset <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msubsup>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>L</mml:mi>
</mml:msubsup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> where <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:msup>
<mml:mi>i</mml:mi>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> training sample <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2286;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>C</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is the corresponding label with <italic>C</italic> classes, and <italic>L</italic> is the number of labeled samples. <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>U</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:msubsup>
<mml:mo>}</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>U</mml:mi>
</mml:mrow>
</mml:msubsup>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> represents a dataset comprising unlabeled samples, where <italic>U</italic> is the number of unlabeled samples. Both <italic>X<sub>L</sub>
</italic> and <italic>X<sub>U</sub>
</italic> share identical semantic labels. The loss function is composed of two terms: <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi>u</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>u</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the supervised loss applied to labeled data, <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>u</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the consistency loss for unlabeled data, and <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi>u</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a scalar hyperparameter.</p>
<p>The supervised loss <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is defined as: <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>B</mml:mi>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>B</mml:mi>
</mml:msubsup>
<mml:mi>H</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>&#x3b7;</italic> denotes the weak augmentation, <italic>B</italic> is the batch size, <italic>H</italic> is the cross-entropy loss, and <italic>p</italic>(&#xb7;) is the output of logits in DNN. Pseudo-labels <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> are generated from weakly augmented unlabeled samples, guiding the prediction of model on strongly augmented samples. The consistency loss <italic>L<sub>u</sub>
</italic> can be formally expressed as: <inline-formula>
<mml:math display="inline" id="im13">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mi>u</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:msubsup>
<mml:mi>II</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im14">
<mml:mi>&#x3d5;</mml:mi>
</mml:math>
</inline-formula> represents strong augmentation, <inline-formula>
<mml:math display="inline" id="im15">
<mml:mi>&#x3bc;</mml:mi>
</mml:math>
</inline-formula> governs the proportion of labeled to unlabeled samples in a minibatch, and II is the indicator function; 0 if the highest probability of unlabeled samples is below the confidence threshold <italic>&#x3c4;</italic> and 1 otherwise.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Wavelet transform</title>
<p>The wavelet transform serves as an effective frequency analysis tool, establishing extensive applications in signal processing (<xref ref-type="bibr" rid="B34">Mallat, 1989</xref>). A wavelet is linked with wavelet and scaling functions, which establish a relationship with the low-pass and high-pass filters to facilitate data decomposition. In practice, the images represent discrete non-stationary signals, involving various frequency intervals and spatial location information. Single-level 2D discrete wavelet transform (<xref ref-type="disp-formula" rid="eq1">Equation 1</xref>) with four filters (<inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>and <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) are often used to decompose an image <italic>x</italic> to obtain its LF component <italic>LF</italic> and three HF components <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>,</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:msub>
<mml:mo>&#x2193;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;&#xa0;</mml:mtext>
<mml:mi>i</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:msub>
<mml:mo>&#x2193;</mml:mo>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the convolution operation with the typical filter <italic>f<sub>i</sub>
</italic> and downsampling operation, receptively. The components acquired through wavelet transform contain distinct information about the raw images of aquatic species (see <xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3A, B</bold>
</xref>). Our method strives to leverage wavelet transform to generate robust information as the input of DNN to extract LF and HF features. As such, a LF entity is represented solely by a LF component (<xref ref-type="disp-formula" rid="eq2">Equation 2</xref>), while a HF entity is represented as a set of HF components in various directions (<xref ref-type="disp-formula" rid="eq3">Equation 3</xref>) similar to those used in (<xref ref-type="bibr" rid="B62">Zhou et&#xa0;al., 2023</xref>):</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Taking FishNet (<xref ref-type="bibr" rid="B20">Khan et&#xa0;al., 2023</xref>) as an example, visualize LF and HF results. <bold>(A)</bold> Raw image. <bold>(B)</bold> Wavelet transform results. <bold>(C)</bold> the HF entity. <bold>(D)</bold> the LF entity.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1373755-g003.tif"/>
</fig>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>F</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>h</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>z</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>v</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>H</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Note that our average HF components aim to reduce computational costs by decreasing the number of subsequent encoders. Ideally, each HF component would be feature-extracted by a specific encoder, but this is computationally expensive. In contrast, our average strategy is orthogonal and complements previous practical approaches for handling HF components, such as element-wise addition (<xref ref-type="bibr" rid="B62">Zhou et&#xa0;al., 2023</xref>), concatenation (<xref ref-type="bibr" rid="B31">Liu et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B10">de Souza Brito et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B29">Liu et&#xa0;al., 2021</xref>), and maximum (<xref ref-type="bibr" rid="B41">Ramamonjisoa et&#xa0;al., 2021</xref>). We refer to Section 4.5 for further details on these.</p>
<p>
<xref ref-type="fig" rid="f3">
<bold>Figures&#xa0;3C, D</bold>
</xref> visually illustrate the LF and HF entities as defined above. By using the LF entity as input, DNN can focus more on LF semantics due to its less noise. In contrast, the HF entity, while exhibiting more noise, offers clearer object boundaries and shapes, enabling DNN to concentrate on HF details. A similar perspective has been adopted by <xref ref-type="bibr" rid="B62">Zhou et&#xa0;al. (2023)</xref> who argues that HF information typically represents image details, while LF information often embodies abstract semantics.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>FusionBlock</title>
<p>Given the entities processed by wavelet transform, we employ parallel encoders equipped with ResNet-50 (<xref ref-type="bibr" rid="B14">He et&#xa0;al., 2016</xref>) to respectively generate high-level LF and HF features. These features are then passed through the FusionBlock to generate attentive features from one stream to another. We argue that applying a cross-stream attention strategy to high-level features can capture the connection between conceptual entities in the LF and HF streams, helping subsequent modules in recognizing aquatic objects in images.</p>
<p>Taking <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> as example to illustrate the details of FusionBlock (where <italic>w</italic>, <italic>h</italic>, <italic>c</italic> and <italic>b</italic> denote width, height, channel number, and batch size), we use a cross-stream attention strategy to explore correlations between the two streams. Specifically, as shown in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>, the two features are passed through four 1 &#xd7; 1 convolutional layers to generate query and key matrices. We reshape the query and key matrices into 3D spatial feature maps (<italic>w</italic> &#xd7; <italic>h</italic> &#xd7; <italic>cb</italic>), and then concatenate them to obtain the fused key and query as (<xref ref-type="disp-formula" rid="eq4">Equations 4</xref>, <xref ref-type="disp-formula" rid="eq5">5</xref>):</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Diagram of the proposed FusionBlock. Here, &#x201c;C&#x201d; signifies feature concatenation, while &#x201c;+&#x201d; represents element-wise addition, &#x201c;&#xd7;&#x201d; denotes dot-product, and &#x201c;M&#x201d; signifies element-wise multiplication.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1373755-g004.tif"/>
</fig>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>q</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>R</italic> denotes the reshape operation, <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>q</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>. After that, the attention map <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mrow>
<mml:mi>w</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>c</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>b</mml:mi>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is computed by performing a dot-product and applying the softmax activation function (<xref ref-type="disp-formula" rid="eq6">Equation 6</xref>).</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>A</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>R</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>q</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:msubsup>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>&#x3c3;</italic> is the softmax activation function. In this way, the feature from one stream could serve to augment another stream. Additionally, to preserve the original information of each stream, a residual connection is employed to fuse the enhanced features with their original counterparts. As such, we obtain the cross-stream attentive features for the two streams as (<xref ref-type="disp-formula" rid="eq7">Equation 7</xref>):</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x24c2;A</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:msub>
<mml:msup>
<mml:mi>f</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>B</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:msub>
<mml:mi>v</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mi>H</mml:mi>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mi>&#x24c2;A</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where &#x24c2; denotes element-wise multiplication, <italic>Bconv</italic>
<sub>1&#xd7;1</sub>(&#xb7;) represents a sequential operation combining a 1 &#xd7; 1 convolutional layer and batch normalization. Once obtaining the cross-stream feature representation, we concatenate these features and apply the dropout operator to the fused feature <italic>f<sub>S</sub>
</italic>. Finally, two fully connected layers are utilized to output the final logits.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Consistency equilibrium loss</title>
<p>In a recent study (<xref ref-type="bibr" rid="B13">Feng et&#xa0;al., 2021</xref>), it was demonstrates that the learning status of a class can be inferred through the mean classification scores. When we take a deeper look into <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1B</bold>
</xref>, it is evident that the head classes exhibit higher mean classification scores, whereas tail classes illustrate lower mean classification scores. Based on this observation, we follow the finding of utilizing the mean classification score to adjust the learning effectiveness of each class throughout the training process. The update process of the mean classification score during training can be illustrated as (<xref ref-type="disp-formula" rid="eq8">Equation 8</xref>):</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>m</mml:mi>
<mml:mo>&#xd7;</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo>+</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>m</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>&#xd7;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mi>C</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> denotes the mean classification score, initialized for each class using <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mi>C</mml:mi>
</mml:mfrac>
<mml:mo>&#xb7;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the mean predicted probability of the sample in a mini-batch, and <italic>m</italic> is a hyper-parameter.</p>
<p>Previous research (<xref ref-type="bibr" rid="B21">Kim et&#xa0;al., 2020</xref>) has revealed that the performance of SSL scheme is highly sensitive to the quality of pseudo-label, and a long-tailed data distribution leads to biased predictions favoring head classes. Utilizing these pseudo-labels in the SSL scheme can be harmful for tail classes. Instead of solely adjusting the class-dependent margin by deriving the mean classification score, the confirmation bias in pseudo-labels should be alleviated at the same time. To this end, we first refine the original pseudo-labels via mean classification score so that match the true data distribution (<xref ref-type="disp-formula" rid="eq9">Equation 9</xref>):</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>a</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>g</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mtext>&#xa0;</mml:mtext>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>&#x3b8;</italic> is a hyper-parameter. Simultaneously, we adaptively adjust the margin by encouraging the tail classes to have larger margins. According to the mean classification score, we add a tunable term to balance the classification, similar to the previous study (<xref ref-type="bibr" rid="B13">Feng et&#xa0;al., 2021</xref>). As such, the CEL can be written as (<xref ref-type="disp-formula" rid="eq10">Equation 10</xref>):</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi>L</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mstyle displaystyle="true">
<mml:munderover>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>B</mml:mi>
</mml:mrow>
</mml:munderover>
<mml:mi>II</mml:mi>
</mml:mstyle>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x3b7;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2265;</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
<mml:mi>H</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x3d5;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>g</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>s</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>We can control the training process through hyper-parameter <italic>&#x3b8;</italic> to ensure the model remains unbiased towards the head classes and does not neglect tail classes. In particular, we increase the larger margin with lower mean classification scores for tail classes, mitigating the suppression of head classes over tail classes to balance the consistency loss.</p>
</sec>
</sec>
<sec id="s4" sec-type="results">
<label>4</label>
<title>Results</title>
<sec id="s4_1">
<label>4.1</label>
<title>Dataset and evaluation metrics</title>
<p>Extensive experiments are conducted using the large-scale FishNet dataset (<xref ref-type="bibr" rid="B20">Khan et&#xa0;al., 2023</xref>), comprising 94,532 images encompassing 17,357 distinct species. Each species is represented by at least one associated image, which span 8 taxonomic classes, 83 orders, 463 families, and 3,826 genera. To validate the effectiveness and universality of our proposed method, we focus on the family classification task. The FishNet dataset categorizes family classes into three groups based on the class frequencies: common, medium, and rare. There are a total of 6 categories in the common group, 52 categories in the medium group, and 405 categories in the rare group. In our experiments, we report the class average accuracy for each group, as well as the overall accuracy over all categories followed as official metrics. The FishNet contains 75,631 images in the training set and 18,901 images in the test set. Unless otherwise stated, we conduct the experiments with a ratio of 20% labeled samples in the training set as labeled data, and the remaining 80% data in the training set as unlabeled data, adhering to the common semi-supervised experimental partition.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Implementation details</title>
<p>We implement our model using PyTorch (<xref ref-type="bibr" rid="B39">Paszke et&#xa0;al., 2019</xref>), with both training and inference procedures conducted on the NVIDIA GeForce RTX 3090 GPU. We use 200 epochs in the training process. In each training step, our batch contains 12 labeled examples and 48 unlabeled examples, maintaining a ratio <italic>&#xb5;</italic> = 4 to support the SSL scheme. To ensure a smooth start, we incorporate linear learning rate warm-up for the first 50 steps, progressively increasing the initial value to 0.004. Subsequently, we decay the learning rate at epochs 30, 60, 100, and 150 by multiplying it by 0.1. For all experiments, the two-stream encoders are initialized with weights pre-trained on the ImageNet dataset (<xref ref-type="bibr" rid="B9">Deng et&#xa0;al., 2009</xref>). We adapt a relatively larger coefficient <italic>m</italic> = 0.99 for the mean classification score update. For the unsupervised loss function CEL, the weight parameter (<italic>&#x3bb;</italic>) increases linearly per epoch according to <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:mi>&#x3bb;</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x3bb;</mml:mi>
<mml:mi>u</mml:mi>
</mml:msub>
<mml:mo>&#xd7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>e</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</inline-formula>, and the confidence threshold <italic>&#x3c4;</italic> is set to 0.95. As in previous works (<xref ref-type="bibr" rid="B48">Sohn et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B22">Lai et&#xa0;al., 2022</xref>), we employ an exponential moving average of model parameters to generate the final performance. We keep other hyper-parameters the same as the ImageNet experiments in FixMatch, except for those mentioned above.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Comparison of aquatic species recognition performance</title>
<p>Several experiments are conducted to elaborate the findings: (a) the baseline utilizing only labeled images for aquatic species recognition based on ResNet-50 (<xref ref-type="bibr" rid="B14">He et&#xa0;al., 2016</xref>); (b) an improved version of (a) incorporating our proposed WFN to enhance image features with wavelet transform; (c) the baseline utilizing both labeled images and unlabeled images based on the representative SSL scheme FixMatch (<xref ref-type="bibr" rid="B48">Sohn et&#xa0;al., 2020</xref>); (d) the proposed WFN integrated into the FixMatch scheme; (e-i) evaluation of state-of-the-art methods designed for long-tailed SSL on the FishNet dataset; (j) utilization of CEL combined with FixMatch; (k) is the final version of our proposed methods incorporating both CEL and WFN into the FixMatch scheme. <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> presents the performance of different methods on the FishNet dataset. Based on these results, several observations emerge regarding the overall progress of the proposed method and variations among different supervised types.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Comparison with supervised, semi-supervised, and long-tailed semi-supervised methods on the FishNet dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center"/>
<th valign="top" align="center">Method</th>
<th valign="top" align="center">SSL</th>
<th valign="top" align="center">LT</th>
<th valign="top" align="center">Common</th>
<th valign="top" align="center">Medium</th>
<th valign="top" align="center">Rare</th>
<th valign="top" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">a)</td>
<td valign="top" align="center">ResNet-50 (<xref ref-type="bibr" rid="B14">He et&#xa0;al., 2016</xref>)</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">70.63</td>
<td valign="top" align="center">57.65</td>
<td valign="top" align="center">21.41</td>
<td valign="top" align="center">26.12</td>
</tr>
<tr>
<td valign="top" align="center">b)</td>
<td valign="top" align="center">WFN</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">72.73</td>
<td valign="top" align="center">58.60</td>
<td valign="top" align="center">25.47</td>
<td valign="top" align="center">29.80</td>
</tr>
<tr>
<td valign="top" align="center">c)</td>
<td valign="top" align="center">FixMatch (<xref ref-type="bibr" rid="B48">Sohn et&#xa0;al., 2020</xref>)</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">79.07</td>
<td valign="top" align="center">64.94</td>
<td valign="top" align="center">22.24</td>
<td valign="top" align="center">27.77</td>
</tr>
<tr>
<td valign="top" align="center">d)</td>
<td valign="top" align="center">FixMatch + WFN</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2013;</td>
<td valign="top" align="center">81.65</td>
<td valign="top" align="center">67.61</td>
<td valign="top" align="center">27.99</td>
<td valign="top" align="center">33.13</td>
</tr>
<tr>
<td valign="top" align="center">e)</td>
<td valign="top" align="center">Fixmatch+CReST (<xref ref-type="bibr" rid="B54">Wei et&#xa0;al., 2021</xref>)</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">68.19</td>
<td valign="top" align="center">67.26</td>
<td valign="top" align="center">24.93</td>
<td valign="top" align="center">30.24</td>
</tr>
<tr>
<td valign="top" align="center">f)</td>
<td valign="top" align="center">Fixmatch+ABC (<xref ref-type="bibr" rid="B25">Lee et&#xa0;al., 2021</xref>)</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">69.14</td>
<td valign="top" align="center">66.71</td>
<td valign="top" align="center">24.98</td>
<td valign="top" align="center">30.24</td>
</tr>
<tr>
<td valign="top" align="center">g)</td>
<td valign="top" align="center">Fixmatch+DARP (<xref ref-type="bibr" rid="B21">Kim et&#xa0;al., 2020</xref>)</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">69.74</td>
<td valign="top" align="center">67.42</td>
<td valign="top" align="center">26.19</td>
<td valign="top" align="center">31.38</td>
</tr>
<tr>
<td valign="top" align="center">h)</td>
<td valign="top" align="center">FixMatch+SAW (<xref ref-type="bibr" rid="B22">Lai et&#xa0;al., 2022</xref>)</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">64.54</td>
<td valign="top" align="center">67.18</td>
<td valign="top" align="center">27.31</td>
<td valign="top" align="center">32.27</td>
</tr>
<tr>
<td valign="top" align="center">i)</td>
<td valign="top" align="center">FixMatch+DASO (<xref ref-type="bibr" rid="B38">Oh et&#xa0;al., 2022</xref>)</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">65.74</td>
<td valign="top" align="center">67.70</td>
<td valign="top" align="center">27.07</td>
<td valign="top" align="center">32.13</td>
</tr>
<tr>
<td valign="top" align="center">j)</td>
<td valign="top" align="center">FixMatch+CEL</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">67.75</td>
<td valign="top" align="center">68.83</td>
<td valign="top" align="center">28.30</td>
<td valign="top" align="center">33.36</td>
</tr>
<tr>
<td valign="top" align="center">k)</td>
<td valign="top" align="center">FixMatch+CEL+WFN</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">69.58</td>
<td valign="top" align="center">68.36</td>
<td valign="top" align="center">32.61</td>
<td valign="top" align="center">37.11</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>From a &#x2192; b, it is evident that the WFN significantly improves overall performance. WFN achieves competitive performance, with average classification accuracy of 72.73%, 58.60%, 25.47%, and 29.80%, surpassing the ResNet-50 by 2.1%, 0.95%, 4.06%, and 3.68% over four metrics. The experiment demonstrates that WFN equipped with wavelet transform and FusionBlock, has better generalization than previous ResNet-50 architecture, which allows the model to tackle the challenges posed by the heterogeneous aquatic environment. From a &#x2192; c, we can observe that the use of SSL yields a notable enhancement compared to the model trained solely using labeled data. The gain from unlabeled data becomes evident in the aquatic species recognition. SSL enables the DL&#xa0;model to leverage the abundance of unlabeled images, further&#xa0;refining its understanding of various species and environmental conditions.</p>
<p>From b &#x2192; d, we can infer a similar conclusion to a &#x2192; c. Furthermore, the combination of WFN and SSL yields a synergistic effect, tackling the challenges posed by the heterogeneity of aquatic environments while leveraging the benefits afforded by unlabeled data. Incorporating WFN into the SSL scheme enables the DL model to acquire robust features across diverse aquatic conditions. In other words, it is crucial to acknowledge that enhanced performance of the robust feature extraction method within SSL extends beyond the initial finding observed in a to c.</p>
<p>
<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> also compares the proposed CEL function with several other methods: CReST (<xref ref-type="bibr" rid="B54">Wei et&#xa0;al., 2021</xref>), which oversample tail classes generated by pseudo-labels, ABC (<xref ref-type="bibr" rid="B25">Lee et&#xa0;al., 2021</xref>), utilizing an auxiliary balanced classifier of a single layer, DARP (<xref ref-type="bibr" rid="B21">Kim et&#xa0;al., 2020</xref>), refining pseudo-labels to match the true distribution of unlabeled data, SAW <xref ref-type="bibr" rid="B22">Lai et&#xa0;al. (2022)</xref>, adjusting weights based on the estimated learning difficulty of each class in unsupervised loss, and DASO (<xref ref-type="bibr" rid="B38">Oh et&#xa0;al., 2022</xref>), employing a blending pseudo-labels strategy to mitigate the overall bias. Since these methods were originally experiment with in the long-tailed SSL domain, we evaluate their performance on the FishNet dataset. We utilize publicly available code to train each method and report the best results obtained from multiple runs, fine-tuning their hyper-parameters to ensure optimal performance. From (e, f, g, h, i) &#x2192; j, we observe that our CEL achieves competitiveness with other methods on the FishNet dataset. From c &#x2192; (e, f, g, h, i, j), the long-tailed extensions yield performance gains of varying degrees for all methods, such as a notable 4.36% increase in average classification accuracy for DASO, demonstrating the importance of long-tailed distribution as a general issue for the task of aquatic species recognition.</p>
<p>Lastly, group (k) demonstrates that integrating WFN and CEL within SSL enhances overall performance for the aquatic species recognition task. The collaborative integration of WFN and CEL could leverage the strengths of each component. WFN enhances the feature extraction capabilities of the model, enabling better handling of the complexities of the aquatic environment. Meanwhile, CEL guides the training process, ensuring that the model benefits from unlabeled data and mitigating long-tailed class imbalanced problems. Through rigorous evaluation, we demonstrate that the combined strength of WFN and CEL contributes to a more robust and accurate aquatic species recognition system, paving the way for advancements in the field of aquatic biodiversity research and conservation.</p>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Ablation study</title>
<sec id="s4_4_1">
<label>4.4.1</label>
<title>Impact of different wavelet bases in WFN</title>
<p>
<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> presents an analysis of various wavelet bases trained on labeled data, including Dmey, Haar, Daubechies 2, Coiflets, Biorthogonal 1.5, and Biorthogonal 2.4. The results we obtained show that the Daubechies 2 wavelet has better classification accuracy, and the Haar wavelet presents better border accuracy. As such, we select the Daubechies 2 wavelet basis as the default for our experiments.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Ablation study for the wavelet bases in WFN.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Wavelet bases</th>
<th valign="top" align="center">Common</th>
<th valign="top" align="center">Medium</th>
<th valign="top" align="center">Rare</th>
<th valign="top" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">Dmey</td>
<td valign="top" align="center">58.86</td>
<td valign="top" align="center">44.36</td>
<td valign="top" align="center">18.48</td>
<td valign="top" align="center">21.91</td>
</tr>
<tr>
<td valign="top" align="center">Haar</td>
<td valign="top" align="center">70.84</td>
<td valign="top" align="center">55.90</td>
<td valign="top" align="center">25.08</td>
<td valign="top" align="center">29.13</td>
</tr>
<tr>
<td valign="top" align="center">Coiflets</td>
<td valign="top" align="center">60.57</td>
<td valign="top" align="center">46.59</td>
<td valign="top" align="center">21.44</td>
<td valign="top" align="center">24.77</td>
</tr>
<tr>
<td valign="top" align="center">Biorthogonal 1.5</td>
<td valign="top" align="center">58.35</td>
<td valign="top" align="center">45.43</td>
<td valign="top" align="center">18.98</td>
<td valign="top" align="center">22.46</td>
</tr>
<tr>
<td valign="top" align="center">Biorthogonal 2.4</td>
<td valign="top" align="center">60.18</td>
<td valign="top" align="center">44.65</td>
<td valign="top" align="center">19.38</td>
<td valign="top" align="center">22.74</td>
</tr>
<tr>
<td valign="top" align="center">Daubechies 2</td>
<td valign="top" align="center">72.73</td>
<td valign="top" align="center">58.60</td>
<td valign="top" align="center">25.47</td>
<td valign="top" align="center">29.80</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_4_2">
<label>4.4.2</label>
<title>Ablation study of different coefficient <italic>&#x3b8;</italic> in CEL</title>
<p>We perform an ablation study on the CEL function with various values of <italic>&#x3b8;</italic> to evaluate the impact of model performance. As shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, an improper proportion of the term, either too large or too small, impedes the attainment of optimal performance. Observing the CEL function reveals a significant variation in the impact of <italic>&#x3b8;</italic>. When the value of <italic>&#x3b8;</italic> is set to 0, the CEL is equivalent to the consistency loss of FixMatch. However, excessively large values of <italic>&#x3b8;</italic> may hinder the ability of model to focus attention on the data, whereas too small values inadequately addresses the bias in long-tailed SSL problem. The trade-off between model performance and CEL when <italic>&#x3b8;</italic> = 0.4 achieves the relatively best performance.</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Ablation study for the hyper-parameter <italic>&#x3b8;</italic> in CEL.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1373755-g005.tif"/>
</fig>
</sec>
<sec id="s4_4_3">
<label>4.4.3</label>
<title>Comparison of fusion strategies for WFN</title>
<p>We further examine the effectiveness of the proposed FusionBlock in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>. We utilize different feature fusion strategies to train the DNNs on the labeled images combined with wavelet transform. The proposed FusionBlock achieves better performance on the FishNet test set compared with element-wise add operation and concatenate features along with channel dimension. We believe that the cross-stream attention fusion strategy is more effective for learning interactive features, making it well-suited for the diverse and challenging environment in aquatic species recognition.</p>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Ablation study for feature fusion strategies in WFN.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Fusion strategy</th>
<th valign="top" align="center">Common</th>
<th valign="top" align="center">Medium</th>
<th valign="top" align="center">Rare</th>
<th valign="top" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">concatenate</td>
<td valign="top" align="center">75.14</td>
<td valign="top" align="center">58.78</td>
<td valign="top" align="center">22.88</td>
<td valign="top" align="center">27.59</td>
</tr>
<tr>
<td valign="top" align="center">Element-wise add</td>
<td valign="top" align="center">72.68</td>
<td valign="top" align="center">58.93</td>
<td valign="top" align="center">23.42</td>
<td valign="top" align="center">28.05</td>
</tr>
<tr>
<td valign="top" align="center">FusionBlock</td>
<td valign="top" align="center">72.73</td>
<td valign="top" align="center">58.60</td>
<td valign="top" align="center">25.47</td>
<td valign="top" align="center">29.80</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Analysis of different frequency components</title>
<p>Since the main semantic information is conveyed in the LF component, previous studies have often used the LF component alone in certain tasks (<xref ref-type="bibr" rid="B26">Li et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B61">Zhao et&#xa0;al., 2023</xref>). However, researchers have attempted to aggregate HF components with methods such as concatenation (<xref ref-type="bibr" rid="B31">Liu et&#xa0;al., 2018</xref>; <xref ref-type="bibr" rid="B10">de Souza Brito et&#xa0;al., 2021</xref>; <xref ref-type="bibr" rid="B29">Liu et&#xa0;al., 2021</xref>), maximum (<xref ref-type="bibr" rid="B41">Ramamonjisoa et&#xa0;al., 2021</xref>), or element-wise addition (<xref ref-type="bibr" rid="B62">Zhou et&#xa0;al., 2023</xref>), and incorporate them into DNNs to improve model performance. To verify the effectiveness of our WFN, we compare the performance of experiments conducted by using different components alone and the ways to connect the HF components. We report the results on FishNet&#x2019;s labeled data in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Analysis of different frequency components.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Raw</th>
<th valign="top" align="center">LF</th>
<th valign="top" align="center">
<italic>HF<sub>Sum</sub>
</italic>
</th>
<th valign="top" align="center">
<italic>HF<sub>Max</sub>
</italic>
</th>
<th valign="top" align="center">
<italic>HF <sub>Concate</sub>
</italic>
</th>
<th valign="top" align="center">
<italic>HF<sub>Average</sub>
</italic>
</th>
<th valign="top" align="center">Common</th>
<th valign="top" align="center">Medium</th>
<th valign="top" align="center">Rare</th>
<th valign="top" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">70.63</td>
<td valign="top" align="center">57.65</td>
<td valign="top" align="center">21.41</td>
<td valign="top" align="center">26.12</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">69.18</td>
<td valign="top" align="center">56.12</td>
<td valign="top" align="center">22.52</td>
<td valign="top" align="center">26.90</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">56.25</td>
<td valign="top" align="center">40.65</td>
<td valign="top" align="center">14.28</td>
<td valign="top" align="center">17.78</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">45.22</td>
<td valign="top" align="center">30.88</td>
<td valign="top" align="center">11.17</td>
<td valign="top" align="center">13.83</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">48.95</td>
<td valign="top" align="center">29.72</td>
<td valign="top" align="center">12.85</td>
<td valign="top" align="center">15.22</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">57.19</td>
<td valign="top" align="center">41.09</td>
<td valign="top" align="center">14.14</td>
<td valign="top" align="center">17.72</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">69.10</td>
<td valign="top" align="center">55.48</td>
<td valign="top" align="center">23.60</td>
<td valign="top" align="center">27.77</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">71.71</td>
<td valign="top" align="center">56.25</td>
<td valign="top" align="center">23.16</td>
<td valign="top" align="center">27.50</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">68.90</td>
<td valign="top" align="center">56.30</td>
<td valign="top" align="center">24.41</td>
<td valign="top" align="center">28.57</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">72.73</td>
<td valign="top" align="center">58.60</td>
<td valign="top" align="center">25.47</td>
<td valign="top" align="center">29.80</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The results show that both HF and LF entities are important for aquatic species recognition, as both HF and LF only attain relatively good performance. From <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>, we find that LF alone achieves better performance than that of only using raw images. One reason for this phenomenon may be the LF entity has less data noise, which enhances the noise-robustness of the DNN by neglecting HF components (<xref ref-type="bibr" rid="B26">Li et&#xa0;al., 2020</xref>). The results also demonstrate despite the noise-robustness in LF leads to quite high performance, the details information conveyed in the HF entity is critical for aquatic species recognition. Furthermore, we compare the strategy of averaging HF components with strategies such as maximum, addition, and concatenation. As illustrated, the model with an averaging connection for HF components used in WFN achieves better performance.</p>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Sensitivity analysis of dataset partition</title>
<p>As shown in <xref ref-type="table" rid="T5">
<bold>Table&#xa0;5</bold>
</xref>, we examine the impact of varying number of labeled and unlabeled data. We set the ratios of labeled data in the training set to 10%, 20%, 30%, and 100%, thereby determining the corresponding ratios of arbitrary unlabeled data. With the entire training dataset labeled (100% labeled data) in supervised learning, the WFN achieves an average classification accuracy of 49.41% across all aquatic species. Furthermore, the overall average classification accuracy of SSL increases by 8.52%, 7.31%, and 10.06% compared to supervised methods when using 10%, 20%, and 30% labeled data and the remaining unlabeled data. Moreover, our method exhibits improved performance with increasing amounts of unlabeled data. Training with 20% labeled data and 40%, 60%, and 80% unlabeled data result in overall average classification accuracy improvements of 5.97%, 6.82%, and 7.31% over the baseline. The empirical results confirm the proficiency of our method in generating pseudo-labels using arbitrary quantities of labeled data. Additionally, the robustness of the proposed method under diverse conditions has been comprehensively validated.</p>
<table-wrap id="T5" position="float">
<label>Table&#xa0;5</label>
<caption>
<p>Sensitivity analysis results of dataset partition strategies.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Labeled</th>
<th valign="top" align="center">Unlabeled</th>
<th valign="top" align="center">Common</th>
<th valign="top" align="center">Medium</th>
<th valign="top" align="center">Rare</th>
<th valign="top" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="center">10%</td>
<td valign="top" align="center">0%<break/>90%</td>
<td valign="top" align="center">66.41<break/>59.78</td>
<td valign="top" align="center">47.35<break/>59.72</td>
<td valign="top" align="center">14.97<break/>23.23</td>
<td valign="top" align="center">19.28<break/>27.80</td>
</tr>
<tr>
<td valign="middle" align="center">20%</td>
<td valign="middle" align="center">0%<break/>40%<break/>60%<break/>80%</td>
<td valign="middle" align="center">72.73<break/>69.27<break/>67.10<break/>69.58</td>
<td valign="middle" align="center">58.60<break/>67.48<break/>68.23<break/>68.36</td>
<td valign="middle" align="center">25.47<break/>31.30<break/>32.11<break/>32.61</td>
<td valign="middle" align="center">29.80<break/>35.77<break/>36.62<break/>37.11</td>
</tr>
<tr>
<td valign="middle" align="center">30%</td>
<td valign="top" align="center">0%<break/>70%</td>
<td valign="top" align="center">75.93<break/>73.05</td>
<td valign="top" align="center">62.35<break/>71.42</td>
<td valign="top" align="center">28.05<break/>38.43</td>
<td valign="top" align="center">32.53<break/>42.59</td>
</tr>
<tr>
<td valign="top" align="center">100%</td>
<td valign="top" align="center">0%</td>
<td valign="top" align="center">83.31</td>
<td valign="top" align="center">74.49</td>
<td valign="top" align="center">45.68</td>
<td valign="top" align="center">49.41</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_7">
<label>4.7</label>
<title>Replacing HF entity with edge information</title>    <p>Wavelet transform and edge detectors such as Canny and Sobel serve similar purposes in extracting detailed information within images. To further demonstrate the effectiveness of the HF entity, we replace the HF entity with the information generated by the edge detector. As shown in <xref ref-type="table" rid="T6">
<bold>Table&#xa0;6</bold>
</xref>, we can see using HF entity outperforms the previous edge detection algorithm by a large margin. To be specific, WFN improves the optimal classification accuracy by over 2.46% in the Canny edge detector, and 1.96% in the Sobel edge detector, respectively. The performance degradation of both experiments illustrates the HF entity extracted by wavelet transform contains rich information about fine details and textures in the image. Besides, Canny and Sobel detectors can be sensitive to noise, especially in low-quality underwater images or those with uneven illumination and complex visual backgrounds, which might lead to false edge detection or noisy information. Through the experiments, we also conclude that WFN has a stronger ability for feature extraction than using raw images with edge information, which can be beneficial for heterogeneous image-collected environments in aquatic species recognition.</p>
<table-wrap id="T6" position="float">
<label>Table&#xa0;6</label>
<caption>
<p>Ablation on effectiveness of various information, including raw image, LF entity, HF entity, and information generated by edge detector.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Raw</th>
<th valign="top" align="center">LF</th>
<th valign="top" align="center">HF</th>
<th valign="top" align="center">Canny</th>
<th valign="top" align="center">Sobel</th>
<th valign="top" align="center">Common</th>
<th valign="top" align="center">Medium</th>
<th valign="top" align="center">Rare</th>
<th valign="top" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">70.63</td>
<td valign="top" align="center">57.65</td>
<td valign="top" align="center">21.41</td>
<td valign="top" align="center">26.12</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">67.92</td>
<td valign="top" align="center">54.55</td>
<td valign="top" align="center">19.56</td>
<td valign="top" align="center">24.11</td>
</tr>
<tr>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">69.66</td>
<td valign="top" align="center">56.01</td>
<td valign="top" align="center">20.27</td>
<td valign="top" align="center">24.93</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center">70.33</td>
<td valign="top" align="center">56.34</td>
<td valign="top" align="center">22.97</td>
<td valign="top" align="center">27.34</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">70.95</td>
<td valign="top" align="center">56.05</td>
<td valign="top" align="center">23.58</td>
<td valign="top" align="center">27.84</td>
</tr>
<tr>
<td valign="top" align="center"/>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center">&#x2713;</td>
<td valign="top" align="center"/>
<td valign="top" align="center"/>
<td valign="top" align="center">72.73</td>
<td valign="top" align="center">58.60</td>
<td valign="top" align="center">25.47</td>
<td valign="top" align="center">29.80</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s4_8">
<label>4.8</label>
<title>Comparison of model size and computation cost</title>
<p>We showcase the performance of models trained on the labeled images along with model size and computational cost. Given that the proposed CEL function is designed for pseudo-labels, its computational complexity is negligible compared to that of fully-supervised training methods. As shown in <xref ref-type="table" rid="T7">
<bold>Table&#xa0;7</bold>
</xref>, WFN requires two encoders for various frequency awareness, which significantly increased the computation cost as the acquired information increased. Furthermore, to illustrate that the performance enhancement stems from well-designed components, we expand ResNet-50 to match the number of parameters and computational costs of WFN. Our results indicate that while increased computational complexity yields positive effects, it still falls shorts of matching the performance of WFN.</p>
<table-wrap id="T7" position="float">
<label>Table&#xa0;7</label>
<caption>
<p>Comparison of model sizes and computational cost.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="center">Method</th>
<th valign="top" align="center">Params (M)</th>
<th valign="top" align="center">Flops (G)</th>
<th valign="top" align="center">Common</th>
<th valign="top" align="center">Medium</th>
<th valign="top" align="center">Rare</th>
<th valign="top" align="center">All</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="center">ResNet-50</td>
<td valign="top" align="center">24.46</td>
<td valign="top" align="center">4.14</td>
<td valign="top" align="center">70.63</td>
<td valign="top" align="center">57.65</td>
<td valign="top" align="center">21.41</td>
<td valign="top" align="center">26.12</td>
</tr>
<tr>
<td valign="top" align="center">ResNet-50<sup>&#x2217;</sup>
</td>
<td valign="top" align="center">59.09</td>
<td valign="top" align="center">11.63</td>
<td valign="top" align="center">72.70</td>
<td valign="top" align="center">57.35</td>
<td valign="top" align="center">23.09</td>
<td valign="top" align="center">27.58</td>
</tr>
<tr>
<td valign="top" align="center">WFN</td>
<td valign="top" align="center">64.11</td>
<td valign="top" align="center">9.11</td>
<td valign="top" align="center">72.73</td>
<td valign="top" align="center">58.60</td>
<td valign="top" align="center">25.47</td>
<td valign="top" align="center">29.80</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>
<sup>&#x2217;</sup> indicates increasing the number of convolutions and channels.</p>
</fn>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec id="s5" sec-type="discussion">
<label>5</label>
<title>Discussion</title>
<p>A previous study (<xref ref-type="bibr" rid="B50">Torney et&#xa0;al., 2019</xref>) demonstrates that recognition task typically requiring four ecologists approximately 3 to 6 weeks for manual analysis can be completed in just 24 hours using DL methods. Their research also concludes that this accelerated approach does not compromise accuracy, as abundance estimates obtained through DL were within 1% of those derived from manual analysis by experts. Computer analysis has the potential to substantially streamline the investigative analysis process. Our study concurs with this point but also underscores the significant challenges in data labeling, as evidenced by previous representative studies (<xref ref-type="bibr" rid="B28">Li et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B43">Rubbens et&#xa0;al., 2023</xref>). This paper introduces a novel technique based on a SSL scheme, where DNN are learned using a limited number of labeled data and extensive unlabeled data, thereby alleviating the burden of manually labeling large dataset for researchers. This is enabled by two simple-to-implement but crucial modifications (1) using a robust feature extraction method, (2) replacing original consistency loss with CEL function. These modifications enable the DL method, trained on a limited amount of labeled data, to effectively address a diverse aquatic&#xa0;environment, as well as the long-tailed distribution of aquatic species.</p>
<p>Aquatic species recognition based on DL serves as a foundation for specific application, particularly biomass estimation and species habitat monitoring (<xref ref-type="bibr" rid="B28">Li et&#xa0;al., 2023</xref>). This information is crucial for informed decision-making in conservation management, including the establishment of protected areas, restoration effort, and mitigation of anthropogenic impacts. Furthermore, our research presents promising applications for long-tailed distribution of aquatic species in the natural world, which can significantly contribute to marine biodiversity conservation efforts. Species distribution and abundance follow a highly skewed rule, with a small number of species exhibiting high abundant, while numerous species are present in relatively low numbers (<xref ref-type="bibr" rid="B52">Villon et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B46">Saleh et&#xa0;al., 2023</xref>). The complex image collection environment poses challenges for commonly used methods such as data augmentation or data generation to be effective, particularly when labeled data is limited. As recommendations for improvement concerning existing conservation measures, we propose integrating our method into established monitoring frameworks. We have conducted both quantitative and qualitative experiments demonstrating the utility of the our method across a variety of diverse aquatic environments using large-scale species recognition datasets. The results of the above experiments instill confidence in our ability to collaborate with existing conservation monitoring programs.</p>
<p>While this work represents progress in developing a robust and effective SSL scheme for real-world aquatic species recognition applications, it has also revealed some limitations that future research should address. Firstly, the CEL enhances the performance of tail-class at the expense of lower performance for head-class. Given the importance of all aquatic species in real environments, it is worthwhile to explore strategies for significantly improving the performance of tail species while maintaining or even enhancing the performance of head species. Secondly, while our study has confirmed the effectiveness of WFN combined with single-level 2D discrete wavelet transform for aquatic species recognition, it is worth developing a DNN equipped with multilevel wavelet packet transform in future research because it could benefit from the hierarchical representation. Lastly, it would be interesting to apply our algorithm to more practical task, such as aquatic species detection, behavior analysis, and trait prediction. By deploying these application in real-world aquatic environments, we can develop increasingly intelligent solutions to address some of the most pressing issues of our time.</p>
</sec>
<sec id="s6" sec-type="conclusions">
<label>6</label>
<title>Conclusion</title>
<p>In this work, we have introduced a robust feature extractor, WFN, and a novel loss function, CEL, based on the SSL scheme FixMatch, for aquatic species recognition. Our proposed methods have demonstrated effectiveness in addressing the challenges of high-quality recognition in complex image-collected environments and the long-tailed class imbalanced nature of aquatic species, even with a limited number of labeled data. This is achieved through dedicated components, using the output of wavelet transform of one to train the DNN, and applying the CEL function at the stage where pseudo-labels come into play. The proposed method has consistently shown performance gains in both quantitative and qualitative experiments. We thus believe that our study can serve as a valuable resource for future research efforts in aquatic species recognition.</p>
</sec>
<sec id="s7" sec-type="data-availability">
<title>Data availability statement</title>
<p>The original contributions presented in the study are included in the article/supplementary material. Further inquiries can be directed to the corresponding authors.</p>
</sec>
<sec id="s8" sec-type="author-contributions">
<title>Author contributions</title>
<p>DM: Writing &#x2013; original draft. JW: Visualization, Writing &#x2013; original draft. LZ: Writing &#x2013; review &amp; editing. FZ: Writing &#x2013; review &amp; editing, Supervision. HW: Writing &#x2013; original draft. XC: Methodology, Writing &#x2013; review &amp; editing. YL: Funding acquisition, Writing &#x2013; original draft. ML: Conceptualization, Writing &#x2013; review &amp; editing.</p>
</sec>
</body>
<back>
<sec id="s9" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the International Research Center of Big Data for&#xa0;Sustainable Development Goals (No. CBAS2022GSP07), and&#xa0;the&#xa0;National Natural Science Foundation of China (No. 42230505, 42206148).</p>
</sec>
<sec id="s10" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s11" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bell</surname> <given-names>K. L. C.</given-names>
</name>
<name>
<surname>Quinzin</surname> <given-names>M. C.</given-names>
</name>
<name>
<surname>Amon</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Poulton</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hope</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sarti</surname> <given-names>O.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Exposing inequities in deep-sea exploration and research: results of the 2022 global deep-sea capacity assessment</article-title>. <source>Front. Mar. Sci</source>. <volume>10</volume>, <elocation-id>1217227</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmars.2023.1217227</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berthelot</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Carlini</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Cubuk</surname> <given-names>E. D.</given-names>
</name>
<name>
<surname>Kurakin</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sohn</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>a). <article-title>Remixmatch: Semi-supervised learning with distribution alignment and augmentation anchoring</article-title>. <source>arXiv preprint arXiv:1911.09785</source>.</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Berthelot</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Carlini</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Goodfellow</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Papernot</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Oliver</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Raffel</surname> <given-names>C. A.</given-names>
</name>
</person-group> (<year>2019</year>b). &#x201c;<article-title>Mixmatch: A holistic approach to semi-supervised learning</article-title>,&#x201d; in <conf-name>Proceedings of the 33rd International Conference on Neural Information Processing Systems (NIPS '19)</conf-name> (<conf-loc>NY, USA</conf-loc>), <volume>454</volume>, <fpage>5049</fpage>&#x2013;<lpage>5059</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5555/3454287.3454741</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname> <given-names>L.</given-names>
</name>
<name>
<surname>McGuire</surname> <given-names>N. E.</given-names>
</name>
<name>
<surname>Hanlon</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Mooney</surname> <given-names>T. A.</given-names>
</name>
<name>
<surname>Girdhar</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Semi-supervised visual tracking of marine animals using autonomous underwater vehicles</article-title>. <source>Int. J. Comput. Vision</source> <volume>131</volume>, <fpage>1406</fpage>&#x2013;<lpage>1427</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11263-023-01762-5</pub-id>
</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cao</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Gaidon</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Arechiga</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Learning imbalanced datasets with label-distribution-aware margin loss</article-title>,&#x201c; in <conf-name>Proceedings of the 33rd International Conference on Neural Information Processing Systems (NIPS'19)</conf-name> (<conf-loc>NY, USA</conf-loc>). <volume>140</volume>, <fpage>1567</fpage>&#x2013;<lpage>1578</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5555/3454287.3454427</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Du</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>X.-D.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.-S.</given-names>
</name>
<name>
<surname>Qian</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Deep-learning-based automated tracking and counting of living plankton in natural aquatic environments</article-title>. <source>Environ. Sci. Technol</source>. <volume>57</volume> (<issue>46</issue>), <fpage>18048</fpage>&#x2013;<lpage>18057</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1021/acs.est.3c00253</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Choi</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Kampffmeyer</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Handegard</surname> <given-names>N. O.</given-names>
</name>
<name>
<surname>Salberg</surname> <given-names>A.-B.</given-names>
</name>
<name>
<surname>Brautaset</surname> <given-names>O.</given-names>
</name>
<name>
<surname>Eikvil</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Semi-supervised target classification in multi-frequency echosounder data</article-title>. <source>ICES J. Mar. Sci.</source> <volume>78</volume>, <fpage>2615</fpage>&#x2013;<lpage>2627</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/icesjms/fsab140</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Cui</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>T.-Y.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Belongie</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2019</year>). &#x201c;<article-title>Class-balanced loss based on effective number of samples</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</conf-name> (<conf-loc>Long Beach, CA, USA</conf-loc>), <fpage>9268</fpage>&#x2013;<lpage>9277</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2019.00949</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Deng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Socher</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>L.-J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Fei-Fei</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2009</year>). &#x201c;<article-title>Imagenet: A large-scale hierarchical image database</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</conf-name> (<conf-loc>Miami, FL, USA</conf-loc>), <fpage>248</fpage>&#x2013;<lpage>255</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2009.5206848</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>de Souza Brito</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Vieira</surname> <given-names>M. B.</given-names>
</name>
<name>
<surname>De Andrade</surname> <given-names>M. L. S. C.</given-names>
</name>
<name>
<surname>Feitosa</surname> <given-names>R. Q.</given-names>
</name>
<name>
<surname>Giraldi</surname> <given-names>G. A.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Combining max-pooling and wavelet pooling strategies for semantic image segmentation</article-title>. <source>Expert Syst. Appl.</source> <volume>183</volume>, <fpage>115403</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2021.115403</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ditria</surname> <given-names>E. M.</given-names>
</name>
<name>
<surname>Lopez-Marcano</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sievers</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Jinks</surname> <given-names>E. L.</given-names>
</name>
<name>
<surname>Brown</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Connolly</surname> <given-names>R. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Automating the analysis of fish abundance using object detection: optimizing animal ecology with deep learning</article-title>. <source>Front. Mar. Sci.</source> <volume>7</volume>, <elocation-id>429</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmars.2020.00429</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Duan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Jiao</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Sar image segmentation based on convolutional-wavelet neural network and markov random field</article-title>. <source>Pattern Recognition</source> <volume>64</volume>, <fpage>255</fpage>&#x2013;<lpage>267</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.patcog.2016.11.015</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Feng</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Exploring classification equilibrium in long-tailed object detection</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF International conference on computer vision (ICCV)</conf-name> (<conf-loc>Montreal, QC, Canada</conf-loc>), <fpage>3417</fpage>&#x2013;<lpage>3426</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV48922.2021.00340</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2016</year>). &#x201c;<article-title>Deep residual learning for image recognition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition (CVPR)</conf-name> (<conf-loc>Las Vegas, NV, USA</conf-loc>), <fpage>770</fpage>&#x2013;<lpage>778</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Huang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>He</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Tan</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Wavelet-srnet: A wavelet-based cnn for multi-scale face super resolution</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE international conference on computer vision (ICCV)</conf-name> (<conf-loc>Venice, Italy</conf-loc>), <fpage>1689</fpage>&#x2013;<lpage>1697</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV.2017.187</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Irfan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Alatawi</surname> <given-names>A. M. M.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Aquatic ecosystem and biodiversity: a review</article-title>. <source>Open J. Ecol.</source> <volume>9</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.4236/oje.2019.91001</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jahanbakht</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Azghadi</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Waltham</surname> <given-names>N. J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Semi-supervised and weakly-supervised deep neural networks and dataset for fish detection in turbid underwater videos</article-title>. <source>Ecol. Inf.</source> <volume>78</volume>, <fpage>102303</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2023.102303</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Katija</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Orenstein</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Schlining</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Lundsten</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Barnard</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Sainz</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Fathomnet: A global image database for enabling artificial intelligence in the ocean</article-title>. <source>Sci. Rep.</source> <volume>12</volume>, <fpage>15914</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-022-19939-2</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kaur</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vijay</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Deep learning with invariant feature based species classification in underwater environments</article-title>. <source>Multimedia Tools Appl.</source>, <fpage>1</fpage>&#x2013;<lpage>22</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042-023-15896-8</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Khan</surname> <given-names>F. F.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Temple</surname> <given-names>A. J.</given-names>
</name>
<name>
<surname>Elhoseiny</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Fishnet: A large-scale dataset and benchmark for fish recognition, detection, and functional trait prediction</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name> (<conf-loc>Paris, France</conf-loc>), <fpage>20496</fpage>&#x2013;<lpage>20506</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV51070.2023.01874</pub-id>
</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kim</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hur</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Park</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Hwang</surname> <given-names>S. J.</given-names>
</name>
<name>
<surname>Shin</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Distribution aligning refinery of pseudo-label for imbalanced semi-supervised learning</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>14567</fpage>&#x2013;<lpage>14579</lpage>.</citation>
</ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Lai</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Gunawan</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Cheung</surname> <given-names>S.-C. S.</given-names>
</name>
<name>
<surname>Chuah</surname> <given-names>C.-N.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Smoothed adaptive weighting for imbalanced semi-supervised learning: Improve reliability against unknown distribution data</article-title>,&#x201d; in <conf-name>International Conference on Machine Learning (PMLR)</conf-name>. <fpage>11828</fpage>&#x2013;<lpage>11843</lpage>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Laradji</surname> <given-names>I. H.</given-names>
</name>
<name>
<surname>Saleh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Rodriguez</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Nowrouzezahrai</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Azghadi</surname> <given-names>M. R.</given-names>
</name>
<name>
<surname>Vazquez</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Weakly supervised underwater fish segmentation using affinity lcfcn</article-title>. <source>Sci. Rep.</source> <volume>11</volume>, <fpage>17379</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-021-96610-2</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>LeCun</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Bengio</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Hinton</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Deep learning</article-title>. <source>nature</source> <volume>521</volume>, <fpage>436</fpage>&#x2013;<lpage>444</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/nature14539</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lee</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Shin</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Abc: Auxiliary balanced classifier for class-imbalanced semisupervised learning</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>34</volume>, <fpage>7082</fpage>&#x2013;<lpage>7094</lpage>.</citation>
</ref>
<ref id="B26">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2020</year>). &#x201c;<article-title>Wavelet integrated cnns for noise-robust image classification</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</conf-name> (<conf-loc>Seattle, WA, USA</conf-loc>), <fpage>7245</fpage>&#x2013;<lpage>7254</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR42600.2020.00727</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Shen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Guo</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Lai</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Wavecnet: Wavelet integrated cnns to suppress aliasing effect for noise-robust image classification</article-title>. <source>IEEE Trans. Image Process.</source> <volume>30</volume>, <fpage>7074</fpage>&#x2013;<lpage>7089</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TIP.2021.3101395</pub-id>
</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Deng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Deep learning for visual recognition and detection of aquatic animals: A review</article-title>. <source>Rev. Aquaculture</source> <volume>15</volume>, <fpage>409</fpage>&#x2013;<lpage>433</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/raq.12726</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A data hiding scheme based on u-net and wavelet transform</article-title>. <source>Knowledge-Based Syst.</source> <volume>223</volume>, <fpage>107022</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.knosys.2021.107022</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Kong</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Qu</surname> <given-names>B.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Fish recognition in the underwater environment using an improved arcface loss for precision aquaculture</article-title>. <source>Fishes</source> <volume>8</volume>, <fpage>591</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/fishes8120591</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Zuo</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2018</year>). &#x201c;<article-title>Multi-level wavelet-cnn for image restoration</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE conference on computer vision and pattern recognition workshops (CVPRW)</conf-name> (<conf-loc>Salt Lake City, UT, USA</conf-loc>), <fpage>773</fpage>&#x2013;<lpage>782</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPRW.2018.00121</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Uemura</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Ge</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>He</surname> <given-names>L.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Fdcnet: filtering deep convolutional network for marine organism classification</article-title>. <source>Multimedia Tools Appl.</source> <volume>77</volume>, <fpage>21847</fpage>&#x2013;<lpage>21860</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11042-017-4585-1</pub-id>
</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ma</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Wei</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Mldet: Towards efficient and accurate deep learning method for marine litter detection</article-title>. <source>Ocean Coast. Manage.</source> <volume>243</volume>, <fpage>106765</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ocecoaman.2023.106765</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mallat</surname> <given-names>S. G.</given-names>
</name>
</person-group> (<year>1989</year>). <article-title>A theory for multiresolution signal decomposition: the wavelet representation</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>11</volume>, <fpage>674</fpage>&#x2013;<lpage>693</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/34.192463</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Menon</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Jayasumana</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Rawat</surname> <given-names>A. S.</given-names>
</name>
<name>
<surname>Jain</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Veit</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Long-tail learning via logit adjustment</article-title>. <source>arXiv preprint arXiv:2007.07314</source>.</citation>
</ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Miyato</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Maeda</surname> <given-names>S.-i.</given-names>
</name>
<name>
<surname>Koyama</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Ishii</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Virtual adversarial training: a regularization method for supervised and semi-supervised learning</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell.</source> <volume>41</volume>, <fpage>1979</fpage>&#x2013;<lpage>1993</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.34</pub-id>
</citation>
</ref>
<ref id="B37">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Moller</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Nilssen</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Nattkemper</surname> <given-names>T. W.</given-names>
</name>
</person-group> (<year>2017</year>). &#x201c;<article-title>Active learning for the classification of species in underwater images from a fixed observatory</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE International Conference on Computer Vision Workshops (ICCVW)</conf-name> (<conf-loc>Venice, Italy</conf-loc>), <fpage>2891</fpage>&#x2013;<lpage>2897</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCVW.2017.341</pub-id>
</citation>
</ref>
<ref id="B38">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Oh</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Kim</surname> <given-names>D.-J.</given-names>
</name>
<name>
<surname>Kweon</surname> <given-names>I. S.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Daso: Distribution-aware semantics-oriented pseudo-label for imbalanced semi-supervised learning</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)</conf-name> (<conf-loc>New Orleans, LA, USA</conf-loc>), <fpage>9786</fpage>&#x2013;<lpage>9796</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR52688.2022.00956</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Paszke</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Gross</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Massa</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Lerer</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Bradbury</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chanan</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Pytorch: An imperative style, high-performance deep learning library</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>32</volume>.</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Qiu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zheng</surname> <given-names>B.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Improving transfer learning and squeeze-and-excitation networks for small-scale fine-grained fish image classification</article-title>. <source>IEEE Access</source> <volume>6</volume>, <fpage>78503</fpage>&#x2013;<lpage>78512</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Access.6287639</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ramamonjisoa</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Firman</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Watson</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lepetit</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Turmukhambetov</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Single image depth prediction with wavelet decomposition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</conf-name> (<conf-loc>Nashville, TN, USA</conf-loc>), <fpage>11089</fpage>&#x2013;<lpage>11098</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01094</pub-id>
</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Ma</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yi</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). &#x201c;<article-title>Balanced meta-softmax for long-tailed visual recognition</article-title>.&#x201c; in <source>Proceedings of the 34th International Conference on Neural Information Processing Systems (NIPS '20)</source> (<conf-loc>NY, USA</conf-loc>), <fpage>4175</fpage>&#x2013;<lpage>4186</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5555/3495724.3496075</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Rubbens</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Brodie</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Cordier</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Destro Barcellos</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Devos</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Fernandes-Salvador</surname> <given-names>J. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Machine learning in marine ecology: an overview of techniques and applications</article-title>. <source>ICES J. Mar. Sci.</source> <volume>80</volume>, <fpage>1829</fpage>&#x2013;<lpage>1853</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/icesjms/fsad100</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sala</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Mayorga</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Bradley</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Cabral</surname> <given-names>R. B.</given-names>
</name>
<name>
<surname>Atwood</surname> <given-names>T. B.</given-names>
</name>
<name>
<surname>Auber</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2021</year>). <article-title>Protecting the global ocean for biodiversity, food and climate</article-title>. <source>Nature</source> <volume>592</volume>, <fpage>397</fpage>&#x2013;<lpage>402</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41586-021-03371-z</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saleh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Laradji</surname> <given-names>I. H.</given-names>
</name>
<name>
<surname>Konovalov</surname> <given-names>D. A.</given-names>
</name>
<name>
<surname>Bradley</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vazquez</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Sheaves</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A realistic fish-habitat dataset to evaluate algorithms for underwater visual analysis</article-title>. <source>Sci. Rep.</source> <volume>10</volume>, <fpage>14671</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-020-71639-x</pub-id>
</citation>
</ref>
<ref id="B46">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saleh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sheaves</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Jerry</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Azghadi</surname> <given-names>M. R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Applications of deep learning in fish habitat monitoring: A tutorial and survey</article-title>. <source>Expert Syst. Appl.</source>, <fpage>121841</fpage>.</citation>
</ref>
<ref id="B47">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Saleh</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sheaves</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Rahimi Azghadi</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Computer vision and deep learning for fish classification in underwater habitats: A survey</article-title>. <source>Fish Fisheries</source> <volume>23</volume>, <fpage>977</fpage>&#x2013;<lpage>999</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/faf.12666</pub-id>
</citation>
</ref>
<ref id="B48">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Sohn</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Berthelot</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Carlini</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Raffel</surname> <given-names>C. A.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Fixmatch: Simplifying semi-supervised learning with consistency and confidence</article-title>. <source>Adv. Neural Inf. Process. Syst.</source> <volume>33</volume>, <fpage>596</fpage>&#x2013;<lpage>608</lpage>.</citation>
</ref>
<ref id="B49">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Tan</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Ouyang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Yin</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). &#x201c;<article-title>Equalization loss for long-tailed object recognition</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</conf-name> (<conf-loc>Seattle, WA, USA</conf-loc>), <fpage>11662</fpage>&#x2013;<lpage>11671</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01168</pub-id>
</citation>
</ref>
<ref id="B50">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Torney</surname> <given-names>C. J.</given-names>
</name>
<name>
<surname>Lloyd-Jones</surname> <given-names>D. J.</given-names>
</name>
<name>
<surname>Chevallier</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Moyer</surname> <given-names>D. C.</given-names>
</name>
<name>
<surname>Maliti</surname> <given-names>H. T.</given-names>
</name>
<name>
<surname>Mwita</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>A comparison of deep learning and citizen science techniques for counting wildlife in aerial survey images</article-title>. <source>Methods Ecol. Evol.</source> <volume>10</volume>, <fpage>779</fpage>&#x2013;<lpage>787</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/2041-210X.13165</pub-id>
</citation>
</ref>
<ref id="B51">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Vaswani</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Shazeer</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Parmar</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Uszkoreit</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Gomez</surname> <given-names>A. N.</given-names>
</name>
<etal/>
</person-group>. (<year>2017</year>). &#x201c;<article-title>Attention is all you need</article-title>,&#x201d; in <conf-name>Proceedings of the 31st International Conference on Neural Information Processing Systems (NIPS'17)</conf-name> (<conf-loc>NY, USA</conf-loc>), <fpage>6000</fpage>&#x2013;<lpage>6010</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5555/3295222.3295349</pub-id>
</citation>
</ref>
<ref id="B52">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Villon</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Iovan</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Mangeas</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Vigliola</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Confronting deep-learning and biodiversity challenges for automatic video-monitoring of marine ecosystems</article-title>. <source>Sensors</source> <volume>22</volume>, <fpage>497</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s22020497</pub-id>
</citation>
</ref>
<ref id="B53">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Visbeck</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Ocean science research is key for a sustainable future</article-title>. <source>Nat. Commun.</source> <volume>9</volume>, <fpage>690</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-018-03158-3</pub-id>
</citation>
</ref>
<ref id="B54">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Wei</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Sohn</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Mellina</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Yuille</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>Crest: A class-rebalancing self-training framework for imbalanced semi-supervised learning</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</conf-name> (<conf-loc>Nashville, TN, USA</conf-loc>), <fpage>10857</fpage>&#x2013;<lpage>10866</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01071</pub-id>
</citation>
</ref>
<ref id="B55">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Hovy</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Luong</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Le</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2020</year>a). &#x201c;<article-title>Unsupervised data augmentation for consistency training</article-title>,&#x201d; in <conf-name>Proceedings of the 34th International Conference on Neural Information Processing Systems (NIPS '20)</conf-name> (<conf-loc>NY, USA</conf-loc>), <fpage>6256</fpage>&#x2013;<lpage>6268</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5555/3495724.3496249</pub-id>
</citation>
</ref>
<ref id="B56">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Xie</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Luong</surname> <given-names>M.-T.</given-names>
</name>
<name>
<surname>Hovy</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Le</surname> <given-names>Q. V.</given-names>
</name>
</person-group> (<year>2020</year>b). &#x201c;<article-title>Self-training with noisy student improves imagenet classification</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition (CVPR)</conf-name> (<conf-loc>Seattle, WA, USA</conf-loc>), <fpage>10687</fpage>&#x2013;<lpage>10698</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CVPR42600.2020</pub-id>
</citation>
</ref>
<ref id="B57">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Yang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>King</surname> <given-names>I.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A survey on deep semi-supervised learning</article-title>. <source>IEEE Trans. Knowledge Data Eng</source>. <volume>35</volume> (<issue>9</issue>), <fpage>8934</fpage>&#x2013;<lpage>8954</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TKDE.2022.3220219</pub-id>
</citation>
</ref>
<ref id="B58">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yao</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Ngo</surname> <given-names>C.-W.</given-names>
</name>
<name>
<surname>Mei</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Wave-vit: Unifying wavelet and transformers for visual representation learning</article-title>,&#x201d; in <conf-name>European Conference on Computer Vision (ECCV)</conf-name> (<conf-loc>Berlin, Heidelberg</conf-loc>), <fpage>328</fpage>&#x2013;<lpage>345</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-3-031-19806-9_19</pub-id>
</citation>
</ref>
<ref id="B59">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Yin</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2021</year>). &#x201c;<article-title>A method for improving accuracy of deeplabv3+ semantic segmentation model based on wavelet transform</article-title>,&#x201d; in <conf-name>International Conference in Communications, Signal Processing, and Systems</conf-name> (<conf-loc>Singapore</conf-loc>), <fpage>315</fpage>&#x2013;<lpage>320</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/978-981-19-0390-8_85</pub-id>
</citation>
</ref>
<ref id="B60">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Kang</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Hooi</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Feng</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Deep long-tailed learning: A survey</article-title>. <source>IEEE Trans. Pattern Anal. Mach. Intell</source>. <volume>45</volume>, <fpage>10795</fpage>&#x2013;<lpage>10816</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TPAMI.2023.3268118</pub-id>
</citation>
</ref>
<ref id="B61">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Wranet: wavelet integrated residual attention u-net network for medical image segmentation</article-title>. <source>Complex intelligent Syst.</source> <volume>9</volume>, <fpage>6971</fpage>&#x2013;<lpage>6983</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s40747-023-01119-y</pub-id>
</citation>
</ref>
<ref id="B62">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). &#x201c;<article-title>Xnet: Wavelet-based low and high frequency fusion networks for fully-and semi-supervised semantic segmentation of biomedical images</article-title>,&#x201d; in <conf-name>Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV)</conf-name> (<conf-loc>Paris, France</conf-loc>), <fpage>21085</fpage>&#x2013;<lpage>21096</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICCV51070.2023.01928</pub-id>
</citation>
</ref>
<ref id="B63">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhuang</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Qiao</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Wildfish++: A comprehensive fish benchmark for multimedia research</article-title>. <source>IEEE Trans. Multimedia</source> <volume>23</volume>, <fpage>3603</fpage>&#x2013;<lpage>3617</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TMM.2020.3028482</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>