<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mar. Sci.</journal-id>
<journal-title>Frontiers in Marine Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mar. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-7745</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmars.2025.1658205</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Marine Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Efficient underwater ecological monitoring with embedded AI: detecting Crown-of-Thorns Starfish via DCGAN and YOLOv6</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Jyothimurugan</surname>
<given-names>Mohan</given-names>
</name>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Pavithra</surname>
<given-names>S.</given-names>
</name>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/3032098/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Deepika Roselind</surname>
<given-names>J.</given-names>
</name>
<uri xlink:href="https://loop.frontiersin.org/people/3186182/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<institution>School of Computer Science and Engineering, Vellore Institute of Technology</institution>, <addr-line>Chennai, Tamil Nadu</addr-line>,&#xa0;<country>India</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/2232062/overview">Katelyn Lawson</ext-link>, Auburn University, United States</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: <ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3124792/overview">Ana Carolina Luz</ext-link>, Instituto de Estudos do Mar Almirante Paulo Moreira, Brazil</p>
<p>
<ext-link ext-link-type="uri" xlink:href="https://loop.frontiersin.org/people/3135715/overview">Ranjith Kumar Dinakaran</ext-link>, Teesside University, United Kingdom</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: S. Pavithra, <email xlink:href="mailto:pavithra.sekar@vit.ac.in">pavithra.sekar@vit.ac.in</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>10</day>
<month>09</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2025</year>
</pub-date>
<volume>12</volume>
<elocation-id>1658205</elocation-id>
<history>
<date date-type="received">
<day>02</day>
<month>07</month>
<year>2025</year>
</date>
<date date-type="accepted">
<day>21</day>
<month>08</month>
<year>2025</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Jyothimurugan, Pavithra and Deepika Roselind.</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Jyothimurugan, Pavithra and Deepika Roselind</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Coral reefs are among the most vital and diverse ecosystems on the planet, providing habitats for marine life, supporting fisheries, and protecting coastlines. However, they are increasingly threatened by outbreaks of Crown-of-Thorns Starfish (COTS), a coral-eating predator capable of causing large-scale reef destruction. Traditional monitoring methods rely on manual diver surveys, which are time-consuming, labour-intensive, and unsuitable for rapid large-scale assessments.</p>
</sec>
<sec>
<title>Methods</title>
<p>To address these limitations, this study proposes an AI-powered framework for detecting COTS in underwater imagery. The system integrates advanced deep learning object detection techniques with synthetic data augmentation to improve model robustness and adaptability under complex underwater conditions. Synthetic training images were generated to expand dataset variability, while optimized detection models were designed for high accuracy and real-time inference.</p>
</sec>
<sec>
<title>Results</title>
<p>The final detection model demonstrated strong performance, achieving a precision of 0.927, recall of 0.903, and mAP@50 of 0.938. These results indicate the effectiveness of the framework in accurately identifying COTS across diverse underwater environments.</p>
</sec>
<sec>
<title>Discussion and Conclusion</title>
<p>The proposed solution is designed for deployment on embedded systems, ensuring practical, scalable, and efficient monitoring of coral reef ecosystems. By enabling real-time and high-accuracy detection of COTS, this framework supports timely interventions and contributes to the conservation and ecological resilience of coral reefs, particularly in vulnerable regions such as the Great Barrier Reef.</p>
</sec>
</abstract>
<kwd-group>
<kwd>Crown-of-Thorns Starfish (COTS)</kwd>
<kwd>underwater object detection</kwd>
<kwd>YOLOv6</kwd>
<kwd>faster R-CNN</kwd>
<kwd>generative adversarial network</kwd>
<kwd>embedded AI</kwd>
<kwd>coral reef monitoring</kwd>
</kwd-group>
<contract-sponsor id="cn001">Vellore Institute of Technology, Chennai<named-content content-type="fundref-id">10.13039/100019904</named-content>
</contract-sponsor>
<counts>
<fig-count count="8"/>
<table-count count="2"/>
<equation-count count="26"/>
<ref-count count="42"/>
<page-count count="17"/>
<word-count count="10436"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Marine Ecosystem Ecology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Coral reefs, often referred to as the &#x201c;rainforests of the sea,&#x201d; are among the most diverse and valuable ecosystems on Earth. They support a vast array of marine life, provide coastal protection, and contribute significantly to the economy through tourism and fisheries. Despite their ecological and economic importance, coral reefs face increasing threats from climate change, pollution, and biological disturbances. One of the most destructive biological threats is the Crown-of-Thorns Starfish (COTS) (Acanthaster planci), a coral-eating echinoderm that can rapidly degrade reef structures by consuming living coral polyps. COTS outbreaks have been recorded across the Indo-Pacific region, with the Great Barrier Reef (GBR) in Australia being one of the most severely affected areas. Each starfish can consume large areas of coral cover, and when populations surge uncontrollably, the resulting damage can outpace coral regeneration. The complexity and vastness of reef ecosystems make manual COTS monitoring challenging, time-consuming, and often inaccurate. Traditional methods involving human divers to visually identify and count COTS are not scalable, provide limited coverage, and are prone to fatigue and bias. These limitations highlight the need for automated, intelligent, and scalable solutions capable of accurately detecting and localizing COTS in real time under diverse underwater conditions.</p>
<p>Existing approaches to underwater object detection, particularly for identifying Crown-of-Thorns Starfish (COTS), face several significant limitations that hinder their applicability in realworld scenarios. One of the primary challenges is environmental complexity underwater imagery is often affected by low contrast, variable lighting conditions, turbidity, and occlusions caused by surrounding marine life such as corals and algae. These factors reduce visual clarity, making it difficult for conventional object detection algorithms to perform effectively. Data scarcity further compounds this challenge. Deep learning models require large volumes of well-labelled data for effective training; however, publicly available annotated datasets for COTS are limited in size and lack diversity. This often leads to overfitting, causing models to perform poorly when applied to different underwater environments. Moreover, high false-positive rates and missed detections remain common in current methods, particularly when class imbalance is not adequately addressed.</p>
<p>Smaller or partially occluded starfish often go undetected, while false alarms increase due to background noise. Another major challenge is the incompatibility of many existing models with real time and embedded deployments. For example, although models like Faster R-CNN achieve high accuracy, their two-stage architecture incurs substantial computational costs, making them inefficient for embedded systems such as underwater drones or remotely operated vehicles (ROVs), where real-time responsiveness and energy efficiency are critical. Additionally, conventional augmentation techniques such as flipping, rotation, and brightness adjustments are insufficient to capture the full diversity and complexity of underwater environments. Consequently, models trained solely with these augmentations often perform poorly in previously unseen conditions, limiting their robustness and adaptability in field applications.</p>
<p>The primary objective of this research is to develop an intelligent, real time, and resource efficient object detection framework for accurately identifying Crown-of-Thorns Starfish (COTS) in complex underwater environments. To achieve this, the study pursues several specific goals. First, it builds a robust detection system by leveraging state-of-the-art deep learning algorithms such as Faster R-CNN and YOLOv6, both adapted to address the unique visual and contextual challenges of underwater imagery. Second, to mitigate the limitations of small and unrepresentative datasets, it employs Deep Convolutional Generative Adversarial Networks (DCGAN) to synthetically generate realistic underwater scenes containing COTS, thereby improving dataset diversity and model generalizability. Third, the models are optimized for deployment on embedded platforms, enabling low latency inference and efficient operation in practical scenarios such as underwater drones or remotely operated vehicles. Finally, model performance is evaluated using standard benchmarking metrics, including Precision, Recall, mAP@50, Inception Score, and Fr&#xe9;chet Inception Distance (FID), with particular emphasis on comparing the effectiveness of real versus GAN generated data in enhancing detection accuracy.</p>
<p>This study introduces a hybrid detection framework that integrates Faster R-CNN, DCGAN, and YOLOv6 to accurately detect Crown-of-Thorns Starfish (COTS) in underwater imagery, achieving a balance between high accuracy and real-time performance. The key innovations are:</p>
<list list-type="bullet">
<list-item>
<p>Hybrid Detection Framework: A novel combination of Faster R-CNN, DCGAN, and YOLOv6 tailored for accurate and efficient COTS detection in diverse underwater environments.</p>
</list-item>
<list-item>
<p>GAN-Based Data Augmentation: Use of DCGAN to generate realistic synthetic underwater images incorporating varied conditions such as turbidity, lighting variations, and occlusions, providing richer diversity than traditional augmentation techniques.</p>
</list-item>
<list-item>
<p>Enhanced Faster R-CNN Architecture: Integration of Res2Net101 backbone with Focal Loss, Triplet Loss, and Soft-NMS to improve detection precision, address class imbalance, and refine bounding boxes in complex marine scenes.</p>
</list-item>
<list-item>
<p>Real-Time Deployment with YOLOv6: YOLOv6, trained on the augmented dataset, is optimized for speed and lightweight deployment on embedded systems such as underwater drones, achieving Precision: 0.927, Recall: 0.903, and mAP@50: 0.938.</p>
</list-item>
<list-item>
<p>Domain-Specific Innovation: The fusion of synthetic data generation, advanced detection architectures, and embedded optimization represents a first-of-its-kind ecological AI solution for marine biodiversity conservation and rapid reef monitoring.</p>
</list-item>
</list>
<p>By combining deep generative modelling, advanced object detection architectures, and real-time optimization, this work delivers a significant advancement in automated marine biodiversity monitoring.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Literature survey</title>
<p>Coral reefs, especially those like the Great Barrier Reef, represent one of the most biologically diverse ecosystems on the planet. However, they are increasingly under threat from climate change, pollution, and biological disturbances, such as the outbreaks of Crown-of-Thorns Starfish (COTS), a coral-eating predator. These outbreaks are difficult to control without early detection, making continuous and large-scale monitoring essential for ecological conservation. Traditional manual survey methods are both labor-intensive and insufficient for large-scale timely assessments. Consequently, researchers have turned to deep learning and embedded artificial intelligence (AI) to provide automated, scalable, and real-time monitoring solutions in underwater environments. This literature review outlines the evolution and convergence of three key domains - underwater object detection, generative adversarial data augmentation, and real-time inference on embedded systems for the development of efficient AI-powered marine monitoring solutions.</p>
<p>Detecting objects in underwater environments introduces unique challenges not encountered in terrestrial domains. Turbidity, poor lighting, backscatter, color distortion, and partial occlusion significantly degrade image quality and object visibility. Standard object detection models must therefore be adapted to the underwater domain for enhanced robustness and reliability. <xref ref-type="bibr" rid="B33">Wang and Xiao (2023)</xref> proposed a highly relevant and domain-specific solution by enhancing the traditional Faster R-CNN architecture. Their improved model incorporated the Res2Net101 backbone for multi-scale feature extraction, coupled with Soft Non-Maximum Suppression (Soft-NMS), Online Hard Example Mining (OHEM) and Generalized Intersection over Union (GIoU) loss. The result was a 3.3% increase in mean Average Precision (mAP), demonstrating the adaptability of region-based detection frameworks in complex marine environments. This work is particularly pertinent as it focuses on small and partially occluded underwater object&#x2019;s conditions under which COTS detection must operate. Similarly, <xref ref-type="bibr" rid="B22">Nambiar and Mittal (2022)</xref> developed a GAN-based super-resolution model tailored for sonar image enhancement, which significantly improved feature visibility in murky environments.</p>
<p>
<xref ref-type="bibr" rid="B21">Lokanath et&#xa0;al. (2017)</xref> also demonstrated the feasibility of Faster R-CNN for object classification and detection in constrained visual conditions. Although not exclusively tailored for underwater scenarios, the robustness of their approach under challenging backgrounds underscores the utility of this architecture as a baseline detector, especially when coupled with domain-specific enhancements. <xref ref-type="bibr" rid="B27">Ren et&#xa0;al.&#x2019;s (2015)</xref> seminal work on Faster R-CNN forms the theoretical foundation for many current object detection models. The introduction of Region Proposal Networks (RPNs) for learning object proposals greatly accelerated and refined the detection pipeline. This architecture, while foundational, has been significantly improved over time to address underwater-specific constraints through advanced backbones and loss functions.</p>
<p>The domain of underwater object detection has gained considerable momentum with the integration
of deep learning techniques, aiming to overcome environmental challenges such as light scattering, low contrast, and visual distortions. Recent studies have focused on combining detection frameworks with image enhancement strategies, notably using Generative Adversarial Networks (GANs), attention mechanisms, and transformer-based models. <xref ref-type="bibr" rid="B19">Liu et&#xa0;al. (2022)</xref> introduced an improved Deep Convolutional GAN (DCGAN) for synthetic image generation, aiding training where annotated underwater data is scarce. Similarly, Thomas et&#xa0;al. (2022) developed a GAN-based super-resolution model tailored for sonar image enhancement, which significantly improved feature visibility in murky environments. These image enhancement efforts serve as a preprocessing step that boosts object detection performance in degraded underwater conditions. <xref ref-type="bibr" rid="B1">Chen and Er (2024)</xref> addressed the challenge of small object detection by proposing Dynamic YOLO, a variant that dynamically adjusts receptive fields based on object size and density. Their method showed superior performance in detecting small-scale marine organisms and submerged objects in cluttered scenes. <xref ref-type="bibr" rid="B2">Chen et&#xa0;al. (2024)</xref> further contributed to data-centric advancements with the WaterPairs dataset, which consists of paired raw and enhanced underwater images. This benchmark supports supervised training for simultaneous image enhancement and object detection.</p>
<p>
<xref ref-type="bibr" rid="B3">Cherian et&#xa0;al. (2022)</xref> provided a comprehensive survey on underwater image enhancement using deep learning, highlighting key architectures like GANs, CNNs, and attention-guided models, while identifying open challenges such as generalization and real-time performance. <xref ref-type="bibr" rid="B4">Dai et&#xa0;al. (2023)</xref> introduced edge-guided representation learning, improving object boundary clarity under blur and low contrast conditions. Their method leverages edge features to guide the learning process, resulting in more precise detection outcomes. <xref ref-type="bibr" rid="B5">Dakhil and Khayeat (2022</xref>; <xref ref-type="bibr" rid="B6">2023</xref>) offered both a review and a methodological framework on underwater object detection using deep learning. Their works emphasize the evolution of detection techniques from classical CNNs to more advanced architectures such as YOLO and Faster R-CNN, advocating for the fusion of enhancement and detection pipelines. <xref ref-type="bibr" rid="B7">Edge et&#xa0;al. (2020)</xref> presented a generative approach that integrates detection cues into image enhancement using GANs. This detection-driven enhancement method produced visually superior images that align with object detection requirements, improving downstream accuracy.</p>
<p>
<xref ref-type="bibr" rid="B9">Fayaz et&#xa0;al. (2024)</xref> developed a joint image restoration and detection model optimized for Autonomous Underwater Vehicles (AUVs), which operates effectively in noisy and low-light underwater scenes. The model&#x2019;s multitask learning approach allows it to simultaneously clean degraded images and detect relevant objects in real-time, supporting AUV-based missions like coral reef inspection and search-and-rescue operations. Collectively, these studies highlight the growing synergy between image enhancement and object detection in underwater vision systems. GAN-based models play a pivotal role in data augmentation and preprocessing, while detection algorithms are evolving toward adaptive, context-aware and lightweight architectures suitable for embedded deployment. Future directions include creating more diverse paired datasets, optimizing transformer based models for real-time use, and building explainable detection frameworks for trustworthy marine exploration.</p>
<p>
<xref ref-type="bibr" rid="B10">Feng and Jin (2024)</xref> proposed CEH-YOLO, a composite-enhanced YOLO model featuring ESPPF and high-order deformable attention modules that improved detection accuracy under conditions of low contrast and blur. <xref ref-type="bibr" rid="B11">Gao et&#xa0;al. (2024)</xref> developed the PE-Transformer, a model incorporating path-enhanced attention mechanisms to better fuse contextual and spatial cues in underwater environments. <xref ref-type="bibr" rid="B12">Guo et&#xa0;al. (2024)</xref> contributed a real-time lightweight detection model by integrating FasterNet into YOLOv8, enabling fast inference and high accuracy on datasets like RUOD and URPC2022. In response to the need for robust datasets, <xref ref-type="bibr" rid="B13">Jian et&#xa0;al. (2024)</xref> conducted a detailed survey of underwater object detection datasets, identifying challenges such as annotation gaps and the need for domain-specific benchmarks.</p>
<p>
<xref ref-type="bibr" rid="B14">Khriss et&#xa0;al. (2024)</xref> explored deep learning strategies specifically for plastic debris detection in marine environments, showcasing how tailored models can address unique underwater ecological problems. <xref ref-type="bibr" rid="B16">Lin et&#xa0;al. (2024)</xref> introduced a detection method that combines learnable query recall with lightweight adapter modules. Their model reduced computational cost while maintaining performance, making it suitable for real-time embedded systems. Similarly, <xref ref-type="bibr" rid="B18">Liu et&#xa0;al. (2024)</xref> presented a lightweight object detection algorithm optimized for embedded deployment, integrating higher order semantic information and image enhancement to improve robustness.</p>
<p>
<xref ref-type="bibr" rid="B17">Liu et&#xa0;al. (2023)</xref> developed TC-YOLO, a model utilizing temporal context and attention mechanisms to improve frame-based detection in underwater videos. Their approach demonstrated consistent detection accuracy in dynamic scenes with moving backgrounds. <xref ref-type="bibr" rid="B20">Liu et&#xa0;al. (2020)</xref> pioneered the use of GANs in combination with YOLOv3 for marine biometric recognition, demonstrating the efficacy of GAN-augmented datasets for improving model training and performance. <xref ref-type="bibr" rid="B21">Lokanath et&#xa0;al. (2017)</xref> laid early groundwork by validating the effectiveness of Faster R-CNN in object classification and detection, showing its adaptability to marine applications despite its computational demands. <xref ref-type="bibr" rid="B23">Nguyen (2022)</xref> presented a practical case study using YOLOv5 with TensorFlow Lite for detecting detrimental starfish on embedded systems. Their approach demonstrated strong performance under constrained resources, enabling real-time monitoring of marine threats. <xref ref-type="bibr" rid="B24">Nooka et&#xa0;al. (2022)</xref> proposed a vision-based deep learning algorithm to detect and track underwater objects. The model combined object detection and tracking capabilities for dynamic underwater surveillance tasks using low-cost camera systems. <xref ref-type="bibr" rid="B25">Pagire et&#xa0;al. (2024)</xref> developed a YOLO-based pipeline for fish detection, with a deep learning model trained on underwater datasets that showed resilience to noisy backgrounds and partial occlusions.</p>
<p>
<xref ref-type="bibr" rid="B34">Wu et&#xa0;al. (2020)</xref> used DCGAN-based data augmentation to enhance detection in agricultural applications. Although focused on plant diseases, their approach inspired similar augmentation strategies in marine detection systems by enhancing training diversity and robustness. <xref ref-type="bibr" rid="B27">Shah et&#xa0;al. (2023)</xref> adopted a zero-shot detection (ZSD) approach for fish recognition in underwater environments, leveraging semantic embeddings to recognize novel species without needing annotated samples. This method offers a scalable solution for biodiversity studies in marine biology.</p>
<p>
<xref ref-type="bibr" rid="B28">Singhal et&#xa0;al. (2025)</xref> analyzed various deep learning architectures, including CNNs, YOLO and Faster R-CNN, in the context of marine exploration. They emphasized cognitive load and energy efficiency, proposing a hybrid framework for deep sea missions with limited bandwidth and hardware constraints. <xref ref-type="bibr" rid="B8">Fang et&#xa0;al. (2018)</xref> previously showed that DCGANs could effectively improve image recognition by generating augmented training data. Their findings remain relevant in underwater domains where data scarcity is a recurring challenge.</p>
<p>
<xref ref-type="bibr" rid="B29">Walia et&#xa0;al. (2024)</xref> explored deep learning techniques for underwater waste detection. Using a customized CNN model, they classified plastic, metal and organic debris with high accuracy and proposed integrating this system into AUVs for automated ocean cleanup missions. <xref ref-type="bibr" rid="B32">Wang H. et al. (2023)</xref> designed a simultaneous restoration and super-resolution GAN model tailored for enhancing underwater images. Their system significantly improved visual quality and clarity, positively impacting detection accuracy in downstream tasks. <xref ref-type="bibr" rid="B33">Wang and Xiao (2023)</xref> further enhanced Faster R-CNN for underwater detection, introducing modifications in ROI pooling and image pre-processing stages. Their approach addressed resolution loss and maintained high accuracy in object localization and classification, making it suitable for coral reef monitoring and underwater inspection.</p>
<p>
<xref ref-type="bibr" rid="B33">Wang and Xiao (2023)</xref> proposed an improved Faster R-CNN framework tailored for underwater environments, addressing image clarity issues through enhanced pre-processing and ROI pooling mechanisms. To improve image quality, <xref ref-type="bibr" rid="B30">Wang J. et&#xa0;al. (2020)</xref> introduced CA-GAN, a class-conditional attention GAN designed for underwater image enhancement, which emphasized object-specific feature recovery and showed improved clarity and contrast in poor-visibility environments. However, <xref ref-type="bibr" rid="B32">Wang Y. et&#xa0;al. (2023)</xref> questioned the assumption that image enhancement alone suffices, presenting a comparative study that showed enhancement can benefit detection, but improvements depend on model and task alignment.</p>
<p>
<xref ref-type="bibr" rid="B37">Zhang F. et&#xa0;al. (2024)</xref> proposed an improved YOLOv8 framework with modifications to the backbone and attention-enhanced layers to increase detection robustness in blurry and low-light underwater conditions. Another contribution by <xref ref-type="bibr" rid="B36">Zhang F. et&#xa0;al. (2023)</xref> focused on YOLOv5 improvements for underwater detection, integrating feature fusion and enhanced anchor box selection. In a follow-up work, <xref ref-type="bibr" rid="B38">Zhang J. et&#xa0;al. (2024)</xref> developed BG-YOLO, a dual-branch system with an image enhancement module guiding detection during training, thus achieving high accuracy without additional inference cost. <xref ref-type="bibr" rid="B38">Zhang J. et&#xa0;al. (2024)</xref> introduced YOLOv7t-CEBC, a compact and efficient detection model specifically designed for underwater litter detection. The network leveraged channel and edge-based components to improve generalization and precision under cluttered backgrounds. <xref ref-type="bibr" rid="B40">Zhao et&#xa0;al. (2024)</xref> explored vision models for environmental monitoring, presenting a hierarchical network for water depth estimation using multi-sensor fusion. Though not a detection model, this work contributes to understanding underwater vision under dynamic environmental conditions.</p>
<p>
<xref ref-type="bibr" rid="B41">Zhou H. et&#xa0;al. (2024)</xref> developed a real-time YOLO-based model optimized for highly complex underwater environments. The model incorporated contrast-aware layers and was evaluated under varying levels of turbidity, highlighting its robustness in real-world deployments. Complementing this, <xref ref-type="bibr" rid="B42">Zhou J. et&#xa0;al. (2024)</xref> presented MFA-CycleGAN, a model for generating sonar images, aiding the training of object detection systems in sonar-based AUVs. Their model contributed to cross-domain generalization by producing synthetic sonar datasets. Finally, <xref ref-type="bibr" rid="B26">Pavithra and Cicil Melbin Denny (2024)</xref> presented the GAN model to showcase the underwater image detection.</p>
<p>These studies demonstrate that effective underwater object detection requires more than selecting a state-of-the-art detector it necessitates tailoring architectures, preprocessing strategies and even image generation techniques to meet environmental demands. While YOLO-based models dominate for real time inference, advancements in GANs and auxiliary tasks such as depth estimation or enhancement continue to elevate overall system performance. Future research should explore multimodal data fusion, domain adaptation for sonar-optical hybrid systems and lightweight training strategies for low-power deployments.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Proposed methodology</title>
<p>To address the challenges of accurate and real-time detection of Crown-of-Thorns Starfish (COTS) in complex underwater environments, this study introduces a hybrid deep learning framework that combines synthetic data generation, high-precision detection, and real-time inference. The proposed methodology integrates a Deep Convolutional Generative Adversarial Network (DCGAN) to enrich training data, an enhanced Faster R-CNN model for accurate object detection, and a lightweight YOLOv6 model optimized for deployment on embedded systems. Each component is designed to complement the others DCGAN mitigates data scarcity, Faster R-CNN ensures robust detection, and YOLOv6 enables real-time performance resulting in a comprehensive solution for automated reef monitoring and conservation are shown in <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>The proposed architecture integrates DCGAN-based data augmentation, enhanced Faster R-CNN object detection, and a lightweight embedded deployment model for real-time Crown-of-Thorns Starfish (COTS) detection in underwater environments.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1658205-g001.tif">
<alt-text content-type="machine-generated">Flowchart showing the process of enhanced object detection using artificial intelligence. It begins with dataset collection and preprocessing, including real and augmented images. This data goes through a generator utilizing transposed convolution, batch normalization, ReLU, and Leaky ReLU. The output progresses to enhanced object detection using a Faster R-CNN, involving an RPN and classification phases. Augmented images are further processed through two &#x201c;Stem&#x201d; phases and multiple &#x201c;CSP Block&#x201d; phases, leading to prediction. The final component is labeled &#x201c;Lab/Cloud Faster-CNN,&#x201d; indicating deployment in field/embedded environments using YOLO V8.</alt-text>
</graphic>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Dataset collection and preprocessing</title>
<p>Underwater image sequences were sourced from various regions of the Great Barrier Reef. These images were captured under a range of environmental conditions, including different lighting levels, water clarity, and background complexities. The dataset includes images showing Crown-of-Thorns Starfish (COTS) at multiple developmental stages such as juvenile, sub-adult, and adult. These images are crucial for understanding the starfish&#x2019;s morphology and behavior in real habitats. Annotation was performed manually by expert marine biologists to ensure high accuracy and domain relevance. Each image was annotated in the Pascal VOC format, which uses XML files to record bounding box coordinates and corresponding class labels. The bounding box is represented by the top-left <inline-formula>
<mml:math display="inline" id="im1">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>n</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mtext>min</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and bottom-right <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mtext>max</mml:mtext>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> coordinates enclosing each starfish. This format supports multi-object annotations per image and is widely supported by object detection frameworks.</p>
<p>The core dataset used comprises annotated underwater images of COTS provided by the CSIRO, enhanced with images sourced from open-access repositories. Given the high intra-class variability and complex underwater textures, preprocessing was applied using histogram equalization, contrast enhancement, and resizing (512&#xd7;512 pixels). This is done to mitigate illumination noise and maximize visual contrast, making feature extraction more robust during training.</p>
<p>Before feeding into deep learning models, all images were resized to a uniform resolution of 512x512 pixels to maintain consistency in input dimensions. Pixel values were then normalized using the standard normalization technique, which involves subtracting the dataset mean and dividing by the standard deviation. This process helps stabilize the model training and accelerates convergence.</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mo>&#x2032;</mml:mo>
</mml:msup>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3bc;</mml:mi>
</mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im3">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula> is original pixel value, <inline-formula>
<mml:math display="inline" id="im4">
<mml:mi>&#x3bc;</mml:mi>
</mml:math>
</inline-formula> is mean of the pixel values in the dataset and <inline-formula>
<mml:math display="inline" id="im5">
<mml:mi>&#x3c3;</mml:mi>
</mml:math>
</inline-formula> is standard deviation of the pixel values in the dataset. This normalization transforms the pixel values to have zero mean and unit variance, which improves the performance of gradient-based optimizers during training. To improve the robustness of the model and reduce overfitting, both traditional and advanced data augmentation techniques were applied.</p>
<sec id="s3_1_1">
<label>3.1.1</label>
<title>Mosaic augmentation</title>
<p>After applying <xref ref-type="disp-formula" rid="eq1">Equation 1</xref>, Mosaic augmentation combines four different training images into one by placing them in a 2x2 grid. This technique enables the model to learn from diverse object scales and locations within a single training instance. It also simulates occlusions and partial visibility, which are common in underwater environments. The transformation for mosaic image combination for each tile <inline-formula>
<mml:math display="inline" id="im6">
<mml:mi>i</mml:mi>
</mml:math>
</inline-formula> can be expressed as:</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im8">
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the position offsets for each quadrant.</p>
</sec>
<sec id="s3_1_2">
<label>3.1.2</label>
<title>Horizontal flipping</title>
<p>After applying <xref ref-type="disp-formula" rid="eq2">Equation 2</xref>, the images are horizontally flipped with a probability of 0.5, effectively doubling the dataset size and enabling the model to generalize better to symmetrical variations of COTS appearances. The flipping transform is provided in <xref ref-type="disp-formula" rid="eq3">Equation 3</xref> as follows.</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>p</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>W</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im9">
<mml:mi>W</mml:mi>
</mml:math>
</inline-formula> is the image width.</p>
</sec>
<sec id="s3_1_3">
<label>3.1.3</label>
<title>Brightness and contrast adjustment</title>
<p>Random brightness and contrast adjustments are applied to simulate varying underwater light conditions caused by depth and particulate matter. The <xref ref-type="disp-formula" rid="eq4">Equation 4</xref> is provided as follows:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mi>a</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>j</mml:mi>
<mml:mi>u</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>d</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>I</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b2;</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im10">
<mml:mi>&#x3b1;</mml:mi>
</mml:math>
</inline-formula> controls contrast and <inline-formula>
<mml:math display="inline" id="im11">
<mml:mi>&#x3b2;</mml:mi>
</mml:math>
</inline-formula>controls brightness.</p>
</sec>
<sec id="s3_1_4">
<label>3.1.4</label>
<title>Gaussian blur</title>
<p>Gaussian blur is applied to replicate the effect of motion blur or lens defocus that occurs in low-visibility underwater scenarios. This helps the model become resilient to slight image degradation. The Gaussian blur kernel is provided as follows:</p>
<disp-formula id="eq5">
<label>(5)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>&#x3c0;</mml:mi>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msup>
<mml:mi>x</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:msup>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:msup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im12">
<mml:mi>&#x3c3;</mml:mi>
</mml:math>
</inline-formula> determines the level of blur. These augmentation techniques used in <xref ref-type="disp-formula" rid="eq5">Equation 5</xref> ensure that the model learns from a wide range of visual scenarios, increasing its ability to perform accurately in real-world, dynamic underwater environments.</p>
</sec>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Synthetic data generation using DCGAN</title>
<p>To address the challenge of limited annotated underwater imagery for Crown-of-Thorns Starfish (COTS), this study employs a Deep Convolutional Generative Adversarial Network (DCGAN) for synthetic data generation. DCGAN was chosen due to its proven ability to generate high-resolution, semantically consistent images in visually cluttered domains like underwater photography. This enriches the training distribution and reduces overfitting.</p>
<p>The architecture shown in <xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> comprise two competing neural networks: a Generator <inline-formula>
<mml:math display="inline" id="im13">
<mml:mi>G</mml:mi>
</mml:math>
</inline-formula> and a Discriminator <inline-formula>
<mml:math display="inline" id="im14">
<mml:mi>D</mml:mi>
</mml:math>
</inline-formula>, trained in a minimax framework. The generator maps a random noise vector <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>~</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> to a synthetic image <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, using a stack of transposed convolutional layers, batch normalization, and non-linear activations (LeakyReLU and Tanh). Meanwhile, the discriminator, built with standard convolutional layers and sigmoid activation, attempts to distinguish real images <inline-formula>
<mml:math display="inline" id="im17">
<mml:mi>x</mml:mi>
</mml:math>
</inline-formula> from generated ones <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Formally, the generator output is denoted as <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>k</mml:mi>
<mml:mi>e</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>z</mml:mi>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, and the discriminator score is computed as <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>&#x3c3;</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>;</mml:mo>
<mml:msub>
<mml:mi>&#x3b8;</mml:mi>
<mml:mi>d</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im21">
<mml:mi>&#x3c3;</mml:mi>
</mml:math>
</inline-formula> is the sigmoid function.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Overview of the deep convolutional generative adversarial network (DCGAN) framework used for generating synthetic Crown-of-Thorns Starfish (COTS) images from real underwater imagery to enhance training diversity and model generalization.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1658205-g002.tif">
<alt-text content-type="machine-generated">Flowchart illustrating a process for improving a crown-of-thorns starfish (COTS) detection model. It starts with real underwater COTS image data, followed by data preprocessing. Then the images are fed into a system using a generator networked with transposed convolution and a discriminator with a convolution layer, creating synthetic images through adversarial training feedback. These synthetic images undergo dataset augmentation. Real and synthetic images are combined to train and improve the COTS detection model, with model evaluation and deployment at the final step.</alt-text>
</graphic>
</fig>
<p>To stabilize training and overcome issues like mode collapse, the model uses Wasserstein GAN with Gradient Penalty (WGAN-GP). The loss function modelled in <xref ref-type="disp-formula" rid="eq6">Equation 6</xref> is based on the Wasserstein-1 (Earth Mover&#x2019;s) distance, formulated as:</p>
<disp-formula id="eq6">
<label>(6)</label>
<mml:math display="block" id="M6">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>~</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>&#xaf;</mml:mo>
</mml:mover>
<mml:mo>~</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:msub>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mrow>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>~</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the real and generated data distributions respectively, <inline-formula>
<mml:math display="inline" id="im24">
<mml:mi>&#x3bb;</mml:mi>
</mml:math>
</inline-formula> is the gradient penalty coefficient (commonly set to 10), and <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
<mml:mi>x</mml:mi>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:mi>&#x3f5;</mml:mi>
<mml:mo>~</mml:mo>
<mml:mi>U</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The gradient penalty term is provided in <xref ref-type="disp-formula" rid="eq7">Equation 7</xref>, <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:msub>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:msub>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mo>&#x2207;</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
</mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mover accent="true">
<mml:mi>x</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> enforces a 1-Lipschitz constraint on the discriminator, which is crucial for stable training. The overall adversarial objective becomes:</p>
<disp-formula id="eq7">
<label>(7)</label>
<mml:math display="block" id="M7">
<mml:mrow>
<mml:munder>
<mml:mrow>
<mml:mi>min</mml:mi>
</mml:mrow>
<mml:mi>G</mml:mi>
</mml:munder>
<mml:munder>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
<mml:mi>D</mml:mi>
</mml:munder>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>G</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im28">
<mml:mi>G</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im29">
<mml:mi>D</mml:mi>
</mml:math>
</inline-formula> are iteratively optimized to produce high-fidelity, realistic images.</p>
<p>The quality and diversity of the generated synthetic images are assessed using two quantitative metrics: Inception Score (IS) and Fr&#xe9;chet Inception Distance (FID). The Inception Score evaluates the clarity and variety of generated images through the Kullback-Leibler divergence between the conditional and marginal label distributions obtained from an Inception v3 network:</p>
<disp-formula id="eq8">
<label>(8)</label>
<mml:math display="block" id="M8">
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>=</mml:mo>
<mml:mtext>exp</mml:mtext>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mi mathvariant="double-struck">E</mml:mi>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>~</mml:mo>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
<mml:mo stretchy="false">[</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>K</mml:mi>
<mml:mi>L</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>|</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>|</mml:mo>
<mml:mo>|</mml:mo>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>)</mml:mo>
<mml:mo stretchy="false">]</mml:mo>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In <xref ref-type="disp-formula" rid="eq8">Equation 8</xref>, a higher IS indicates better quality and more class diversity in the generated images. On the other hand, in <xref ref-type="disp-formula" rid="eq9">Equation 9</xref>, FID quantifies the distributional similarity between real and generated images in the feature space of an Inception network. It is calculated as:</p>
<disp-formula id="eq9">
<label>(9)</label>
<mml:math display="block" id="M9">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3bc;</mml:mi>
<mml:mi>g</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mi>T</mml:mi>
<mml:mi>r</mml:mi>
<mml:mo stretchy="false">(</mml:mo>
<mml:msub>
<mml:mtext>&#x3a3;</mml:mtext>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mtext>&#x3a3;</mml:mtext>
<mml:mi>g</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mtext>&#x3a3;</mml:mtext>
<mml:mi>r</mml:mi>
</mml:msub>
<mml:msub>
<mml:mtext>&#x3a3;</mml:mtext>
<mml:mi>g</mml:mi>
</mml:msub>
<mml:msup>
<mml:mo stretchy="false">)</mml:mo>
<mml:mrow>
<mml:mfrac bevelled="true">
<mml:mi>1</mml:mi>
<mml:mn>2</mml:mn>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Where <inline-formula>
<mml:math display="inline" id="im30">
<mml:mi>&#x3bc;</mml:mi>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math display="inline" id="im31">
<mml:mtext>&#x3a3;</mml:mtext>
</mml:math>
</inline-formula> are the means and covariances of the feature activations for real <inline-formula>
<mml:math display="inline" id="im32">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>r</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and generated <inline-formula>
<mml:math display="inline" id="im33">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>g</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> samples.</p>
<p>In the context of this work, the DCGAN achieved a high Inception Score of approximately 1152.07 and a low FID of 3.79, confirming that the synthetic images were visually convincing and statistically similar to real underwater data. These results demonstrate that DCGAN-generated data can effectively enrich the training set, enhancing the performance and generalizability of downstream object detection models in complex marine environments.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Enhanced real-time object detection using YOLOv6 and hybrid pipeline</title>
<sec id="s3_3_1">
<label>3.3.1</label>
<title>Enhanced faster R-CNN for underwater COTS detection</title>
<p>To improve Crown-of-Thorns Starfish (COTS) detection under challenging underwater conditions, the standard Faster R-CNN architecture is enhanced with a Res2Net101 backbone and advanced loss and optimization techniques. Faster R-CNN consists of two-stage detector: a Region Proposal Network (RPN) that generates candidate object regions and a detection head that performs classification and bounding box regression.</p>
<p>Replacing the backbone with Res2Net101 improves multi-scale feature extraction. Res2Net101 embeds hierarchical residual-like connections within a single residual block, splitting the feature map into smaller groups, each processed with different receptive fields:</p>
<disp-formula id="eq10">
<label>(10)</label>
<mml:math display="block" id="M10">
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>s</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>s</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im34">
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> operates on a feature subset <inline-formula>
<mml:math display="inline" id="im35">
<mml:mrow>
<mml:msub>
<mml:mi>x</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, allowing the network to learn from both fine and coarse details. <xref ref-type="disp-formula" rid="eq10">Equation 10</xref> is especially helpful in detecting partially occluded or small COTS instances.</p>
</sec>
<sec id="s3_3_2">
<label>3.3.2</label>
<title>Loss functions and optimization</title>
<p>To address underwater-specific challenges such as class imbalance, occlusion, and overlapping bounding boxes, three loss functions are integrated. The focal loss provided in <xref ref-type="disp-formula" rid="eq11">Equation 11</xref> addresses class imbalance by emphasizing hard-to-classify samples. Triplet Loss in <xref ref-type="disp-formula" rid="eq12">Equation 12</xref> promotes better feature embedding separation between starfish and background. The dual-loss combination enhances the classifier&#x2019;s ability to discern subtle features amidst background noise. To handle underwater challenges such as class imbalance, occlusion, and bounding box overlap, several enhancements were introduced. The focal loss is specified as:</p>
<disp-formula id="eq11">
<label>(11)</label>
<mml:math display="block" id="M11">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mi>&#x3b3;</mml:mi>
</mml:msup>
<mml:mtext>log</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>This loss down-weights easy examples and focuses the model on hard, misclassified samples. Here, <inline-formula>
<mml:math display="inline" id="im36">
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the predicted probability of the true class, <inline-formula>
<mml:math display="inline" id="im37">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3b1;</mml:mi>
<mml:mi>t</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is a weighting term and <inline-formula>
<mml:math display="inline" id="im38">
<mml:mi>&#x3b3;</mml:mi>
</mml:math>
</inline-formula> is the focusing parameter. The Triplet loss is specified as,</p>
<disp-formula id="eq12">
<label>(12)</label>
<mml:math display="block" id="M12">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>max</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>&#x2016;</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>a</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>f</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>&#x2016;</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:mo>+</mml:mo>
<mml:mi>&#x3b1;</mml:mi>
<mml:mo>,</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>This helps in improving feature space clustering by minimizing the distance between an anchor and a positive (same class) and maximizing it from a negative (different class). The Generalized Intersection over Union (GIoU) provided in <xref ref-type="disp-formula" rid="eq4">Equation 13</xref>,</p>
<disp-formula id="eq13">
<label>(13)</label>
<mml:math display="block" id="M13">
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>U</mml:mi>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im39">
<mml:mi>C</mml:mi>
</mml:math>
</inline-formula> is the area of the smallest enclosing box covering both the prediction and ground truth, and <inline-formula>
<mml:math display="inline" id="im40">
<mml:mi>U</mml:mi>
</mml:math>
</inline-formula>
<bold>is their union.</bold>
</p>
</sec>
<sec id="s3_3_3">
<label>3.3.3</label>
<title>Loss functions and optimization</title>
<p>Non-Maximum Suppression with Soft-NMS reduces missed detections in overlapping regions, improving detection of clustered starfish. Soft-NMS is chosen over conventional NMS since it often discards overlapping true positives whereas soft-NMS lowers confidence scores rather than eliminating them entirely, thus improving recall. GIoU provides better gradient feedback when there is no overlap. The equation for Soft-NMS is as follows,</p>
<disp-formula id="eq14">
<label>(14)</label>
<mml:math display="block" id="M14">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#xb7;</mml:mo>
<mml:msup>
<mml:mi>e</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>I</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>U</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
<mml:mi>&#x3c3;</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In <xref ref-type="disp-formula" rid="eq14">Equation 14</xref>, instead of suppressing overlapping boxes entirely, Soft-NMS reduces their confidence scores smoothly, preserving useful predictions when objects are close together or partially overlapping. This enhanced Faster R-CNN is well-suited for visually complex reef settings with murky water, variable lighting, and cluttered backgrounds. Res2Net101 boosts fine-detail recognition and multi-scale detection. Focal and triplet loss improves classification robustness in imbalanced and noisy conditions. Finally, GIoU and Soft-NMS increases localization accuracy and recall in overlapping and occluded cases.</p>
</sec>
<sec id="s3_3_4">
<label>3.3.4</label>
<title>YOLOv6 architecture and training</title>
<p>YOLOv6 is a high-speed, single-stage object detector optimized for edge devices. It integrates feature extraction, classification and bounding box regression in a single pass, reducing inference latency compared to two-stage methods. In <xref ref-type="disp-formula" rid="eq15">Equation 15</xref>, the architecture comprises of the backbone which contains CSPDarkNet for semantic feature extraction, PANet for multi-scale feature aggregation and Anchor-free detection head for final predictions. Each prediction is,</p>
<disp-formula id="eq15">
<label>(15)</label>
<mml:math display="block" id="M15">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>h</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im41">
<mml:mrow>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> denotes the bounding box centre, <inline-formula>
<mml:math display="inline" id="im42">
<mml:mi>w</mml:mi>
</mml:math>
</inline-formula>, <inline-formula>
<mml:math display="inline" id="im43">
<mml:mi>h</mml:mi>
</mml:math>
</inline-formula> is the width, height and <inline-formula>
<mml:math display="inline" id="im44">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the class confidence scores. The loss function used for training combines three components: object loss (binary cross entropy) in <xref ref-type="disp-formula" rid="eq16">Equation 16</xref>, classification loss (cross-entropy) in <xref ref-type="disp-formula" rid="eq17">Equation 17</xref> and localization loss in <xref ref-type="disp-formula" rid="eq18">Equation 18</xref> using Complete IoU (CIoU). These are defined as,</p>
<disp-formula id="eq16">
<label>(16)</label>
<mml:math display="block" id="M16">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>j</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mo stretchy="false">[</mml:mo>
<mml:mtext>y&#xa0;log</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>y</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mtext>log</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>p</mml:mtext>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq17">
<label>(17)</label>
<mml:math display="block" id="M17">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mstyle displaystyle="true">
<mml:msubsup>
<mml:mo>&#x2211;</mml:mo>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mi>c</mml:mi>
</mml:msubsup>
<mml:mrow>
<mml:msub>
<mml:mi>y</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mi>log</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mstyle>
</mml:mrow>
</mml:math>
</disp-formula>
<disp-formula id="eq18">
<label>(18)</label>
<mml:math display="block" id="M18">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>l</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mtext>CIoU</mml:mtext>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mtext>B</mml:mtext>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>B</mml:mi>
<mml:mrow>
<mml:mi>g</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>The CIoU term enhances bounding box regression by considering overlap area, centre distance and aspect ratio. YOLOv6 was trained using a hybrid dataset combining real underwater images and synthetic samples generated by DCGAN. This enhances the model&#x2019;s ability to generalize across varying reef environments. Data augmentations such as random scaling, blurring, color jitter and cut-mix simulate real-world underwater distortions. To optimize for embedded systems such as Jetson Nano, Coral TPU or NVIDIA Xavier, post-training quantization (e.g., INT8) and model pruning are applied. Inference is accelerated using ONNX Runtime or TensorRT, enabling detection in milliseconds per frame. This makes YOLOv6 suitable for deployment in underwater robots, reef monitoring stations or real-time drone feeds where low power and fast response are critical.</p>
</sec>
<sec id="s3_3_5">
<label>3.3.5</label>
<title>Dataset enhancement with DCGAN</title>
<p>To improve robustness in varied reef environments, the dataset merges real underwater images with synthetic samples generated by a Deep Convolutional GAN (DCGAN) using <xref ref-type="disp-formula" rid="eq19">Equation 19</xref>. The synthetic dataset is modelled:</p>
<disp-formula id="eq19">
<label>(19)</label>
<mml:math display="block" id="M19">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mo>{</mml:mo>
<mml:mi>G</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>|</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo>&#x223c;</mml:mo>
<mml:msub>
<mml:mi>p</mml:mi>
<mml:mi>z</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>z</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>and the full training dataset is:</p>
<disp-formula id="eq20">
<label>(20)</label>
<mml:math display="block" id="M20">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x222a;</mml:mo>
<mml:mo>&#x200b;</mml:mo>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>y</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>t</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In <xref ref-type="disp-formula" rid="eq20">Equation 20</xref>, data augmentation includes random scaling, blurring, colour jitter and cut-mix to replicate underwater distortions.</p>
</sec>
<sec id="s3_3_6">
<label>3.3.6</label>
<title>Hybrid detection pipeline</title>
<p>Both YOLOv6 and Faster R-CNN are trained on the same enriched dataset. This enriched dataset is then used to train both Faster R-CNN and YOLOv6 models. Faster R-CNN, with its two-stage architecture, provides highly accurate bounding boxes and class predictions. It uses a Region Proposal Network (RPN) that generates proposals <inline-formula>
<mml:math display="inline" id="im45">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> based on anchor boxes and filters them via Non-Maximum Suppression (NMS). The final detection loss for Faster R-CNN provided in <xref ref-type="disp-formula" rid="eq21">Equation 21</xref> is a combination of classification and bounding box regression losses.</p>
<disp-formula id="eq21">
<label>(21)</label>
<mml:math display="block" id="M21">
<mml:mrow>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3bb;</mml:mi>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mi>b</mml:mi>
<mml:mi>b</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>x</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</disp-formula>
<p>YOLOv6, in contrast, is optimized for speed and performs detection in a single forward pass. Its architecture uses an anchor-free head and outputs a tensor.</p>
<disp-formula id="eq22">
<label>(22)</label>
<mml:math display="block" id="M22">
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">[</mml:mo>
<mml:mrow>
<mml:mi>B</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">]</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>B</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>w</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>,</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>w</mml:mi>
<mml:mi>h</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>f</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>d</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>n</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>e</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>S</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Using <xref ref-type="disp-formula" rid="eq22">Equation 22</xref>, this makes it suitable for real-time applications on edge devices. Each model in the hybrid system has a defined operational role. YOLOv6 deployed on embedded devices (Jetson Nano, Coral TPU and NVIDIA Xavier) with post-training quantization (INT8) and model pruning. Accelerated with ONNX Runtime or TensorRT for millisecond-scale inference. Faster R-CNN is used in batch processing or cloud-based analysis for high-precision validation. DCGAN improves training diversity by introducing edge cases and rare instances.</p>
</sec>
<sec id="s3_3_7">
<label>3.3.7</label>
<title>Dynamic model selection</title>
<p>A scheduler or control logic can dynamically switch between models based on the operational context (e.g., latency threshold or available compute resources). In real-world deployments, a trade-off exists between detection accuracy and computational efficiency. Using <xref ref-type="disp-formula" rid="eq23">Equation 23</xref>, a scheduler selects the model based on accuracy (<inline-formula>
<mml:math display="inline" id="im46">
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) and inference time (<inline-formula>
<mml:math display="inline" id="im47">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), the system defines an optimization constraint as,</p>
<disp-formula id="eq23">
<label>(23)</label>
<mml:math display="block" id="M23">
<mml:mrow>
<mml:mi>M</mml:mi>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mtext>arg&#xa0;max</mml:mtext>
</mml:mrow>
<mml:mrow>
<mml:mi>m</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mi>R</mml:mi>
<mml:mi>C</mml:mi>
<mml:mi>N</mml:mi>
<mml:mi>N</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>Y</mml:mi>
<mml:mi>O</mml:mi>
<mml:mi>L</mml:mi>
<mml:mi>O</mml:mi>
<mml:mi>v</mml:mi>
<mml:mn>6</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:munder>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mi>A</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>&#x3f5;</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xa0;</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <inline-formula>
<mml:math display="inline" id="im48">
<mml:mi>&#x3f5;</mml:mi>
</mml:math>
</inline-formula> is a small constant to avoid division by zero and accounts for practical timing constraints. This hybrid approach delivers scalable, adaptive monitoring balancing YOLOv6&#x2019;s speed for in-field deployment with Faster R-CNN&#x2019;s accuracy for offline analysis, while DCGAN ensures resilience to diverse underwater conditions.</p>
</sec>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experimental results and discussions</title>
<p>The dataset used in this study comprises both real and synthetic images of Crown-of-Thorns Starfish (COTS). The primary real-world dataset employed is the CSIRO Crown-of-Thorns Starfish Detection Dataset, publicly available via the arXiv repository. This dataset includes high-resolution underwater images collected from diverse reef zones of the Great Barrier Reef, showcasing COTS in different life stages and under varying environmental conditions such as depth, turbidity, coral density and lighting. Each image is accompanied by detailed annotations in Pascal VOC XML format, specifying bounding boxes for individual COTS instances. The dataset provides a robust foundation for supervised learning tasks, enabling both detection and classification.</p>
<p>A total of 2,437 annotated images are used from the CSIRO dataset, with an average resolution of 1280&#xd7;720 pixels. The annotation covers multiple visual conditions are murky water, overlapping organisms, coral camouflage and the dataset is manually verified. To further enhance data variability and overcome class imbalance, synthetic images are generated using a Deep Convolutional Generative Adversarial Network (DCGAN). The synthetic dataset simulates realistic underwater artifacts and lighting distortions to complement the real images. The final training set comprises a 70:30 split of real and synthetic images, further divided into 80:20 for training and validation. Images are pre-processed through normalization, resizing to 512&#xd7;512 pixels and augmented using techniques like mosaic augmentation, horizontal flipping and Gaussian noise. This hybrid dataset forms the backbone for training both YOLOv6 and Faster R-CNN models used in the system.</p>
<sec id="s4_1">
<label>4.1</label>
<title>End-to-end pipeline integration</title>
<p>The proposed system adopts a modular, end-to-end architecture for the automated detection of Crown-of-Thorns Starfish (COTS) in underwater ecosystems. It begins with the acquisition of both real and synthetic images. The primary real dataset used is the CSIRO Crown-of-Thorns Starfish Detection Dataset, comprising high-resolution underwater images captured from various locations along the Great Barrier Reef. These images include manual annotations marking different life stages of COTS. The synthetic dataset is generated using a Deep Convolutional Generative Adversarial Network (DCGAN), which enhances the diversity of training samples by mimicking varied underwater conditions such as turbidity, lighting, and occlusion.</p>
<p>The network architecture is composed of three key components: the DCGAN module, which generates 13,000 synthetic training images by learning the distribution of the real dataset; the Faster R-CNN module, which is a two-stage detector enhanced with a Res2Net101 backbone and advanced loss functions (Focal Loss, Triplet Loss, GIoU) for high-accuracy detection and evaluation; and the YOLOv6 module, which is a single-stage detector optimized for real-time inference on edge devices, employing CIoU loss and anchor-free heads for efficient localization.</p>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Training and testing</title>
<p>Model training is conducted on the combined dataset of real and synthetic images using standardized deep learning practices. Faster R-CNN is trained using cross-entropy loss for classification and GIoU loss for bounding box regression, optimized using stochastic gradient descent (SGD) with momentum. YOLOv6 is trained using a compound loss consisting of object, classification, and CIoU localization components, optimized using the Adam optimizer. Training is performed over 100 to 150 epochs with a batch size of 16. The models are validated using an 80:20 train-test split with k-fold cross-validation to ensure generalizability.</p>
<p>Extensive data augmentation, including resizing, rotation, flipping, blurring and synthetic data integration is applied to improve model robustness. Evaluation is performed using key metrics such as precision, recall, f1-score, mAP@50 and mAP@50:95 with results visualized using bounding box overlays and precision-recall curve plots. The machine learning implementation is carried out using PyTorch and TensorFlow frameworks. The pipeline includes modular training scripts for each model component, checkpoint saving, hyperparameter scheduling, and TensorBoard based real-time visualization of training loss and metric curves. A separate preprocessing pipeline handles real-time normalization, resizing and bounding box encoding. The training loop includes gradient clipping, dynamic learning rate reduction (Reduce LROn Plateau) and early stopping for convergence efficiency. Custom callbacks are implemented for tracking per-class accuracy, IoU score distribution, and GPU utilization logs. The models are trained on NVIDIA GPUs using mixed precision (FP16) training for memory and compute efficiency with training workflows containerized via Docker to support portability and reproducibility.</p>
<p>For deployment, Faster R-CNN is reserved for centralized lab environments where high computational resources are available, supporting in-depth analysis and model retraining. YOLOv6, due to its lightweight architecture, is deployed on edge devices such as Jetson Nano for real-time inference in reef environments. Trained models are exported to ONNX format and further optimized using TensorRT to accelerate inference. A lightweight web-based dashboard, built with Flask or Node.js, provides a monitoring interface to visualize detections, system logs, and field outputs. This integrated architecture balances high-precision offline analysis with real-time, scalable deployment, offering a practical and intelligent tool for coral reef monitoring and marine conservation.</p>
</sec>
<sec id="s4_3">
<label>4.3</label>
<title>Evaluation metrics</title>
<p>To evaluate the effectiveness of the proposed hybrid detection system for Crown-of-Thorns Starfish (COTS), a comprehensive set of evaluation metrics was employed. For classification tasks, the key performance indicators included precision, recall, f1-score and mean average precision (mAP) at various IoU thresholds. In <xref ref-type="disp-formula" rid="eq24">Equation 24</xref>, precision measures the proportion of correctly identified COTS among all instances predicted as COTS and is defined as:</p>
<disp-formula id="eq24">
<label>(24)</label>
<mml:math display="block" id="M24">
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In <xref ref-type="disp-formula" rid="eq25">Equation 25</xref>, recall quantifies the proportion of actual COTS instances that were correctly detected.</p>
<disp-formula id="eq25">
<label>(25)</label>
<mml:math display="block" id="M25">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>T</mml:mi>
<mml:mi>P</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>F</mml:mi>
<mml:mi>N</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>In <xref ref-type="disp-formula" rid="eq26">Equation 26</xref>, the f1-score, which combines both precision and recall, is given by,</p>
<disp-formula id="eq26">
<label>(26)</label>
<mml:math display="block" id="M26">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>=</mml:mo>
<mml:mn>2</mml:mn>
<mml:mo>&#xb7;</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>&#xa0;</mml:mo>
<mml:mo>&#xb7;</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>P</mml:mi>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>o</mml:mi>
<mml:mi>n</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>R</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>c</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Detection performance was further evaluated using mAP@50 and mAP@50:95, which assess the accuracy of bounding box predictions at a fixed IoU threshold of 0.5 and a range from 0.5 to 0.95, respectively. For evaluating the quality of synthetic images generated by the DCGAN, two standard generative metrics were used. The Inception Score (IS) provided in <xref ref-type="disp-formula" rid="eq8">Equation 8</xref> evaluates image quality and diversity by computing the KL divergence between the conditional label distribution <inline-formula>
<mml:math display="inline" id="im49">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:mi>y</mml:mi>
<mml:mo>|</mml:mo>
<mml:mi>x</mml:mi>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and the marginal distribution <inline-formula>
<mml:math display="inline" id="im50">
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>y</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Meanwhile, the Fr&#xe9;chet Inception Distance (FID) provided in <xref ref-type="disp-formula" rid="eq9">Equation 9</xref> compares the statistics (mean <italic>&#x3bc;</italic> and covariance &#x3a3;) of real and generated image features in the Inception v3 model&#x2019;s latent space. The DCGAN achieved an Inception Score of 3.82 and an FID of 3.79, indicating that the synthetic images were visually coherent and statistically close to real underwater COTS scenes.</p>
<p>
<xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> demonstrate the evolution of training and validation loss over 20 epochs. The training loss (green curve) exhibits a consistent and smooth downward trend, starting from approximately 0.36 and steadily decreasing to 0.22. This indicates that the model is effectively learning the training data and optimizing its weights. The validation loss (orange curve), on the other hand, fluctuates more notably in the early epochs. It begins at 0.25, peaks around epoch 3 (~0.30), and then gradually stabilizes between epochs 8 to 15, before slightly declining toward the end, reaching around 0.263. This stabilization after early volatility suggests that the model generalizes well and is not overfitting.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Loss trends for training and validation datasets over 20 epochs, demonstrating model convergence and generalization behavior during the learning process.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1658205-g003.tif">
<alt-text content-type="machine-generated">Line graph showing training loss and validation loss over 20 epochs. The training loss, in green, decreases steadily from 0.36 to 0.22. The validation loss, in orange, starts at 0.26, peaks at 0.3, and then gradually decreases to 0.27.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref> visualizes the adversarial training dynamics of a DCGAN model: yellow line (G) which represents the Generator Loss and orange line (D) which represents the Discriminator Loss. The generator loss fluctuates significantly between 1.0 to 3.0, with frequent spikes. These variations indicate that the generator is constantly adapting to fool the discriminator by producing increasingly realistic synthetic images. The high variance is expected in adversarial training, especially as the generator is trying to explore the latent space effectively. A consistently high generator loss may also reflect that the discriminator is doing a good job at identifying fake samples. The discriminator loss remains relatively stable and low, typically ranging from 0.1 to 0.7. This suggests that the discriminator confidently distinguishes between real and synthetic images during most epochs. However, periodic increases in discriminator loss indicate that the generator occasionally succeeds in confusing it, which is a sign of a healthy adversarial competition.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Visualization of the adversarial training dynamics between the generator (G) and discriminator (D) in the DCGAN model used for augmenting underwater Crown-of-Thorns Starfish imagery.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1658205-g004.tif">
<alt-text content-type="machine-generated">Line graph displaying generator (blue line) and discriminator (orange line) loss during training over iterations. The generator loss fluctuates significantly, peaking around 25, while the discriminator loss remains low, mostly under 5.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4_4">
<label>4.4</label>
<title>Comparison with benchmark models</title>
<p>
<xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> shows that the Faster R-CNN, enhanced with the Res2Net101 backbone and loss functions like Focal Loss, Triplet Loss, and GIoU, yielded excellent detection results: a precision of 0.946, recall of 0.917, F1-score of 0.931, mAP@50 of 0.945 and mAP@50:95 of 0.872. These metrics confirm its robustness in complex underwater scenes, especially under conditions involving occlusion and class imbalance. The lightweight YOLOv6, trained on the same hybrid dataset, achieved a precision of 0.927, recall of 0.903, F1-score of 0.915, and mAP@50 of 0.938. Impressively, it operated at an average inference speed of ~28 milliseconds per frame on an NVIDIA Jetson Nano, validating its suitability for real-time, edge-level deployments such as underwater drones and autonomous reef monitors and star fish detection are shown in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Comparative evaluation of faster R-CNN and YOLOv6 for COTS detection.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Metric</th>
<th valign="middle" align="center">Faster R-CNN</th>
<th valign="middle" align="center">YOLOv6</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">Precision</td>
<td valign="middle" align="center">0.946</td>
<td valign="middle" align="center">0.927</td>
</tr>
<tr>
<td valign="middle" align="left">Recall</td>
<td valign="middle" align="center">0.917</td>
<td valign="middle" align="center">0.903</td>
</tr>
<tr>
<td valign="middle" align="left">F1-Score</td>
<td valign="middle" align="center">0.931</td>
<td valign="middle" align="center">0.915</td>
</tr>
<tr>
<td valign="middle" align="left">mAP@50</td>
<td valign="middle" align="center">0.945</td>
<td valign="middle" align="center">0.938</td>
</tr>
<tr>
<td valign="middle" align="left">mAP@50:95</td>
<td valign="middle" align="center">0.872</td>
<td valign="middle" align="center">&#x2013;</td>
</tr>
<tr>
<td valign="middle" align="left">Inference Time (ms/frame)</td>
<td valign="middle" align="center">~120 ms/frame</td>
<td valign="middle" align="center">~28ms/frame</td>
</tr>
<tr>
<td valign="middle" align="left">Deployment</td>
<td valign="middle" align="center">High-accuracy, lab/validation</td>
<td valign="middle" align="center">Real-time, edge deployment</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>Inception Score and FID Score for Image Synthesis (DCGAN).</p>
</fn>
<fn>
<p>Inception Score:</p>
</fn>
<fn>
<p>Mean, 1152.0713006897054.</p>
</fn>
<fn>
<p>Std, 1.730163492944744.</p>
</fn>
<fn>
<p>FID Score: 3.7880779876580704.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Visual comparison between ground-truth labels and predicted bounding boxes for Crown-of-Thorns Starfish (COTS) on underwater validation images using the proposed GAN-augmented hybrid Faster R-CNN architecture.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1658205-g005.tif">
<alt-text content-type="machine-generated">Underwater images of coral reefs with each panel labeled by file name. Some panels have red boxes indicating the presence of starfish, labeled as &#x201c;starfish.&#x201d; The background water is a clear blue, highlighting the reef structures.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> present two confusion matrices highlighting the performance of the proposed models. The left panel shows the confusion matrix for the YOLOv6 detection model, indicating robust performance with 48 true negatives, 44 true positives, 5 false positives, and 3 false negatives.</p>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Comparison of the detection accuracy of the YOLOv6 model (left) and classification performance of the DCGAN-augmented starfish recognition module (right), evaluated on the validation dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1658205-g006.tif">
<alt-text content-type="machine-generated">Two side-by-side matrices. The first is a YOLOv6 confusion matrix with true labels for detection and no detection against predicted labels. Values: 48 true negatives, 5 false positives, 3 false negatives, 44 true positives. The second is a DCGAN-based starfish classification matrix with true values for starfish and background against predicted values. Values: 0.89 starfish accuracy, 1.0 background accuracy. Both use color gradients from light to dark blue to represent values.</alt-text>
</graphic>
</fig>
<p>This demonstrates YOLOv6&#x2019;s strong capability in distinguishing between COTS and non-COTS instances in real-time scenarios. The right panel shows the normalized confusion matrix for a DCGAN-enhanced classifier distinguishing between starfish and background classes. The classifier correctly identifies 89% of starfish instances and 100% of background samples, validating the effectiveness of GAN-augmented training data in improving fine-grained marine object classification. The image synthesis results obtained using DCGAN yielded an Inception Score with a mean of 1152.07 &#xb1; 1.73, and a FID Score of 3.79, indicating high-quality and diverse generated images.</p>
<p>Positive results are obtained when a DCGAN model is evaluated for picture synthesis. With a standard deviation of 1.73 and a mean Inception Score of 1152.07, the model shows that it can produce varied and high-quality images. Furthermore, obtaining a FID Score of 3.79 indicates a noteworthy similarity between generated and genuine images, underscoring the model&#x2019;s effectiveness in generating realistic results. These results highlight the DCGAN framework&#x2019;s effectiveness and demonstrate how it may be used to advance picture synthesis jobs. However, the inference time for Faster R-CNN is relatively high at ~120 ms/frame, which restricts its use to offline or centralized lab-based validation setups where computational resources are abundant and latency is less critical. In contrast, YOLOv6 offers a more balanced trade-off between accuracy and real-time performance. With a precision of 0.927, recall of 0.903, and F1-score of 0.915, it is only marginally behind Faster R-CNN in detection accuracy. Its mAP@50 is 0.938, which is highly competitive, though mAP@50:95 was not evaluated in this deployment. Crucially, YOLOv6 runs at ~28 ms/frame, making it suitable for real-time inference on embedded systems.</p>
<p>
<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> illustrates the variation of three key performance metrics Precision, Recall, and Mean Average Precision at IoU 0.5 (mAP@50) with respect to increasing Intersection-over-Union (IoU) thresholds. The Precision-IOU curve (red) remains high across lower IoU thresholds and begins to drop sharply beyond 0.7, reflecting a decline in exact localization accuracy. The Recall-IOU curve (blue) shows a relatively stable behavior until 0.6 before gradually decreasing. The mAP@50 curve (purple) demonstrates the overall robustness of the model, maintaining consistency across moderate threshold.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Evaluation of the hybrid model&#x2019;s detection performance across varying IoU thresholds, showcasing the Precision-IoU, Recall-IoU, and mAP@50-IoU curves.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1658205-g007.tif">
<alt-text content-type="machine-generated">Line graph titled &#x201c;Metrics v IOU Thresholds&#x201d; with IOU Threshold on the x-axis and Score on the y-axis. It features three curves: Precision-IOU (red), Recall-IOU (blue), and mAP50-IOU (purple). All curves start near a score of 1 and trend downward as IOU Threshold increases from 0 to 1.</alt-text>
</graphic>
</fig>
<p>
<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> show the benchmark comparison of various object detection models. With a precision of 0.927 and mAP@50 of 0.938, YOLOv6 demonstrates exceptional accuracy while achieving real-time performance at ~28 ms/frame on Jetson Nano. It uses anchor-free heads and an optimized backbone tailored for embedded systems, making it ideal for real-time COTS detection in underwater drones and field-deployable units. The most accurate model in the comparison, with a precision of 0.946 and F1-score of 0.931. The use of Res2Net101 backbone and loss functions like Focal and GIoU enables robustness in occluded or complex reef conditions. However, its inference time (~120ms/frame) makes it more suitable for lab-based validation or offline batch processing.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Benchmarking of object detection models for COTS identification in underwater environments.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Model</th>
<th valign="middle" align="center">Dataset</th>
<th valign="middle" align="center">Backbone</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">F1-Score</th>
<th valign="middle" align="center">mAP@50</th>
<th valign="middle" align="center">Inference Time (ms/frame)</th>
<th valign="middle" align="center">Remarks</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="middle" align="left">YOLOv6 (Proposed)</td>
<td valign="middle" align="left">CSIRO + DCGAN</td>
<td valign="middle" align="left">Custom YOLOv6</td>
<td valign="middle" align="center">0.927</td>
<td valign="middle" align="center">0.903</td>
<td valign="middle" align="center">0.915</td>
<td valign="middle" align="center">0.938</td>
<td valign="middle" align="center">~28</td>
<td valign="middle" align="left">High-speed and accurate; ideal for edge deployment</td>
</tr>
<tr>
<td valign="middle" align="left">Faster R-CNN (Proposed)</td>
<td valign="middle" align="left">CSIRO + DCGAN</td>
<td valign="middle" align="left">Res2Net101</td>
<td valign="middle" align="center">0.946</td>
<td valign="middle" align="center">0.917</td>
<td valign="middle" align="center">0.931</td>
<td valign="middle" align="center">0.945</td>
<td valign="middle" align="center">~120</td>
<td valign="middle" align="left">High precision and robustness; suitable for lab validation</td>
</tr>
<tr>
<td valign="middle" align="left">YOLOv5s (<xref ref-type="bibr" rid="B31">Wang H. et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B33">Wang and Xiao, 2023</xref>)</td>
<td valign="middle" align="left">MS COCO</td>
<td valign="middle" align="left">CSPDarkNet</td>
<td valign="middle" align="center">0.908</td>
<td valign="middle" align="center">0.885</td>
<td valign="middle" align="center">0.896</td>
<td valign="middle" align="center">0.921</td>
<td valign="middle" align="center">~35</td>
<td valign="middle" align="left">Efficient on standard datasets, but slightly lower accuracy on underwater</td>
</tr>
<tr>
<td valign="middle" align="left">YOLOv4 (<xref ref-type="bibr" rid="B21">Lokanath et&#xa0;al., 2017</xref>)</td>
<td valign="middle" align="left">MS COCO</td>
<td valign="middle" align="left">CSPDarkNet53</td>
<td valign="middle" align="center">0.913</td>
<td valign="middle" align="center">0.882</td>
<td valign="middle" align="center">0.897</td>
<td valign="middle" align="center">0.925</td>
<td valign="middle" align="center">~32</td>
<td valign="middle" align="left">Strong accuracy but bulkier model size</td>
</tr>
<tr>
<td valign="middle" align="left">SSD (<xref ref-type="bibr" rid="B34">Wu et&#xa0;al., 2020</xref>)</td>
<td valign="middle" align="left">PASCAL VOC</td>
<td valign="middle" align="left">VGG16</td>
<td valign="middle" align="center">0.841</td>
<td valign="middle" align="center">0.799</td>
<td valign="middle" align="center">0.819</td>
<td valign="middle" align="center">0.823</td>
<td valign="middle" align="center">~45</td>
<td valign="middle" align="left">Lightweight, fast; suffers under challenging underwater scenes</td>
</tr>
<tr>
<td valign="middle" align="left">RetinaNet (<xref ref-type="bibr" rid="B8">Fang et&#xa0;al., 2018</xref>)</td>
<td valign="middle" align="left">MS COCO</td>
<td valign="middle" align="left">ResNet50</td>
<td valign="middle" align="center">0.879</td>
<td valign="middle" align="center">0.857</td>
<td valign="middle" align="center">0.868</td>
<td valign="middle" align="center">0.899</td>
<td valign="middle" align="center">~75</td>
<td valign="middle" align="left">Balanced recall but slower than YOLO series</td>
</tr>
<tr>
<td valign="middle" align="left">EfficientDet-D1 (<xref ref-type="bibr" rid="B19">Liu et&#xa0;al., 2022</xref>)</td>
<td valign="middle" align="left">MS COCO</td>
<td valign="middle" align="left">EfficientNet-B1</td>
<td valign="middle" align="center">0.886</td>
<td valign="middle" align="center">0.849</td>
<td valign="middle" align="center">0.867</td>
<td valign="middle" align="center">0.903</td>
<td valign="middle" align="center">~55</td>
<td valign="middle" align="left">Strong balance; performance drops under low contrast scenes</td>
</tr>
<tr>
<td valign="middle" align="left">YOLOv3 (<xref ref-type="bibr" rid="B40">Zhao et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="left">MS COCO</td>
<td valign="middle" align="left">Darknet-53</td>
<td valign="middle" align="center">0.897</td>
<td valign="middle" align="center">0.870</td>
<td valign="middle" align="center">0.883</td>
<td valign="middle" align="center">0.913</td>
<td valign="middle" align="center">~33</td>
<td valign="middle" align="left">Outdated but still effective baseline</td>
</tr>
<tr>
<td valign="middle" align="left">Faster R-CNN (<xref ref-type="bibr" rid="B23">Nguyen, 2022</xref>)</td>
<td valign="middle" align="left">PASCAL VOC</td>
<td valign="middle" align="left">ResNet101</td>
<td valign="middle" align="center">0.892</td>
<td valign="middle" align="center">0.860</td>
<td valign="middle" align="center">0.876</td>
<td valign="middle" align="center">0.910</td>
<td valign="middle" align="center">~135</td>
<td valign="middle" align="left">Accurate but slower inference for real time tasks</td>
</tr>
<tr>
<td valign="middle" align="left">DETR (<xref ref-type="bibr" rid="B15">Li et&#xa0;al., 2024</xref>)</td>
<td valign="middle" align="left">MS COCO</td>
<td valign="middle" align="left">Transformer</td>
<td valign="middle" align="center">0.901</td>
<td valign="middle" align="center">0.867</td>
<td valign="middle" align="center">0.884</td>
<td valign="middle" align="center">0.918</td>
<td valign="middle" align="center">~100</td>
<td valign="middle" align="left">Transformer-based model; slower but powerful on structured scenes</td>
</tr>
<tr>
<td valign="middle" align="left">CenterNet (<xref ref-type="bibr" rid="B35">Xu et&#xa0;al., 2023</xref>)</td>
<td valign="middle" align="left">MS COCO</td>
<td valign="middle" align="left">Hourglass-104</td>
<td valign="middle" align="center">0.874</td>
<td valign="middle" align="center">0.841</td>
<td valign="middle" align="center">0.857</td>
<td valign="middle" align="center">0.890</td>
<td valign="middle" align="center">~40</td>
<td valign="middle" align="left">Heatmap based keypoint detection; moderate speed and accuracy</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>A lightweight and highly popular model that achieves decent precision (0.908) and speed (~35 ms/frame). While it performs well on MS COCO, it slightly underperforms on underwater datasets due to domain shift and less emphasis on small object detection. A reliable upgrade from YOLOv3 with better accuracy (mAP@50 = 0.925) and a decent speed of ~32 ms/frame. Its heavier backbone, CSPDarkNet53, improves depth but makes it less agile for edge deployment. Once popular for real-time object detection, SSD offers faster inference (~45 ms/frame) but with significantly lower precision (0.841) and recall (0.799), particularly under challenging conditions like turbidity or coral occlusion, which are common in underwater environments. Introducing Focal Loss, RetinaNet achieves a fair balance with 0.879 precision and 0.868 F1-score. However, it&#x2019;s slower (~75 ms/frame) and has difficulty detecting multiple overlapping or small-scale objects efficiently. This model strikes a solid balance between accuracy (0.903 mAP@50) and model efficiency. It leverages compound scaling and EfficientNet-B1 as a backbone, which makes it suitable for mobile GPUs. However, in underwater datasets with low contrast, performance tends to degrade.</p>
<p>Despite being older, YOLOv3 maintains relevance with an F1-score of 0.883 and a decent mAP@50 of 0.913 are shown in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>. It is still used as a baseline in many applications but lacks architectural innovations like those in YOLOv4&#x2013;v6.Using a ResNet101 backbone, this version is precise (0.892) but has a long inference time (~135 ms/frame). It&#x2019;s not suitable for embedded systems but performs well in controlled high-resource environments. A novel transformer-based approach achieving a strong F1-score (0.884) and mAP@50 (0.918), DETR excels in structured scenes but suffers from high computational demand (~100 ms/frame) and a slow convergence rate during training, making it less practical for on-the-fly reef monitoring. This keypoint based object detection model offers moderate performance (F1 = 0.857) and speed (~40 ms/frame). While it handles object localization innovatively, it may miss detections in cluttered scenes due to reliance on centre point estimation.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Comparative performance of object detection models across Precision, Recall, F1-Score and mAP@50.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-12-1658205-g008.tif">
<alt-text content-type="machine-generated">Bar chart comparing model performance across various detection models including YOLOv6, Faster R-CNN, and others. Metrics such as precision, recall, F1-score, and mAP@50 are represented by red, blue, green, and yellow bars. Most models score above 0.8 for each metric.</alt-text>
</graphic>
</fig>
</sec>
<sec id="s4_5">
<label>4.5</label>
<title>Failure case analysis</title>
<p>Despite the high precision and real-time detection performance achieved by the proposed hybrid deep learning framework, certain limitations were observed, particularly under extreme underwater conditions. One of the most significant challenges arises in scenes affected by heavy turbidity or poor illumination. These conditions, common in deeper or sediment-rich reef zones, substantially degrade image contrast and visibility, making it difficult for both the Faster R-CNN and YOLOv6 models to differentiate starfish from background clutter. As a result, the models occasionally fail to generate bounding boxes around the Crown-of-Thorns Starfish (COTS), leading to false negatives or mislocalized predictions.</p>
<p>Another notable failure case involves partial occlusions, where COTS are hidden behind coral branches or overlapping with other marine structures. In such instances, the Region Proposal Network (RPN) in Faster R-CNN fails to isolate complete object features, often resulting in either incomplete bounding boxes or misclassification as background elements. Moreover, despite the integration of synthetic images through DCGAN augmentation, certain coral structures with similar radial textures or color palettes continue to be misclassified as starfish. This background confusion is particularly evident in complex reef scenes where visually similar marine organisms (e.g., sea cucumbers or branching corals) are mistakenly detected as COTS with moderate confidence scores ranging between 0.5 and 0.7.</p>
<p>As highlighted in <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>, qualitative predictions show that while most starfish are detected accurately, some predictions fail due to low confidence or imprecise localization. In cluttered reef environments, YOLOv6 occasionally generates overlapping or redundant bounding boxes with low confidence, affecting the overall mAP@50:95 scores. These observations underline the importance of addressing real world underwater variability in future work. To mitigate these limitations, future enhancements will involve training with more diverse and adversarial augmented samples using Wasserstein GAN with Gradient Penalty (WGAN-GP) to simulate extreme underwater degradation more realistically. By integrating attention based modules such as the Convolutional Block Attention Module (CBAM) or transformer-based encoders can help focus on relevant spatial features even under occlusion or camouflage. Domain adaptation techniques will also be explored to improve model generalizability across varying reef ecosystems and sensor settings. Overall, while the current system demonstrates strong performance under standard conditions, acknowledging and addressing these failure scenarios is vital for deploying reliable ecological monitoring systems in diverse and dynamic marine environments.</p>
</sec>
<sec id="s4_6">
<label>4.6</label>
<title>Computational complexity</title>
<p>The computational complexity of the proposed and benchmarked models varies significantly based on their architecture, number of parameters, memory footprint, and inference speed. YOLOv6, the proposed real-time model, comprises approximately 37 million parameters and requires around 95 GFLOPs (Giga Floating Point Operations) per inference. Its lightweight design, coupled with anchor-free heads and optimization for TensorRT deployment, results in a low memory footprint (~250 MB) and a fast inference speed of ~28 milliseconds per frame, making it highly suitable for embedded systems such as Jetson Nano and reef-side underwater drones. On the other hand, Faster R-CNN, while delivering top-tier accuracy, has a significantly higher computational load. With around 140 million parameters and over 206 GFLOPs, it demands approximately 700 MB of memory and has an average inference time of ~120 milliseconds per frame. This makes it ideal for centralized lab-based analysis or post-processing tasks where computational power is not a limiting factor, but unsuitable for real-time embedded applications. Among other models, YOLOv5s stands out with just 7.2 million parameters and only 16.5 GFLOPs, resulting in a highly compact memory usage (~90 MB) and real-time inference at ~35 ms/frame. It is extremely well-suited for low-power edge deployments, though slightly less accurate than YOLOv6. YOLOv4 strikes a balance between performance and complexity, with ~64 million parameters and ~90 GFLOPs, offering solid accuracy and moderate hardware demands.</p>
<p>RetinaNet, despite offering a good F1-Score through the use of Focal Loss, incurs ~97 GFLOPs and has a higher inference delay (~75 ms/frame) due to its dense predictions and deeper backbone. EfficientDet-D1, while using only ~6 million parameters and ~2.5 GFLOPs, is extremely efficient in both parameter count and memory usage (~70 MB), making it suitable for mobile and low power scenarios, though it may underperform in low contrast underwater imagery. Legacy models like YOLOv3 remain competitive with ~61.5 million parameters and ~66 GFLOPs, maintaining ~33 ms/frame inference. However, newer architectures like DETR, a Transformer-based model, come with a trade-off of higher complexity (around 41 million parameters, ~86 GFLOPs, and ~100 ms/frame) and longer training times. CenterNet, with ~52 million parameters, relies on keypoint estimation and offers moderate complexity (~96 GFLOPs) and ~40 ms/frame inference speed, but struggles in densely packed or occluded environments. In summary, YOLOv6 offers the best trade-off between speed and accuracy, whereas Faster R-CNN remains superior in precision but is computationally expensive. Models like YOLOv5s and EfficientDet-D1 offer alternatives for ultra-low-power deployment, while newer architectures such as DETR promise high accuracy at the cost of slower inference and higher resource demands.</p>
</sec>
<sec id="s4_7" sec-type="discussion">
<label>4.7</label>
<title>Discussion, limitation, and conclusion</title>
<sec id="s4_7_1" sec-type="discussion">
<label>4.7.1</label>
<title>Discussion</title>
<p>Compared to conventional COTS monitoring methods, such as surveys, the proposed AI-based approach offers substantial advantages in spatial coverage, temporal frequency, and scalability. Traditional surveys are constrained by human endurance, occupational safety considerations, and environmental conditions, which limit both the area surveyed and the frequency of data collection. In contrast, the automated system can operate continuously, acquire large-scale datasets, and process information in near real time, thereby facilitating earlier detection of infestation events.</p>
<p>The system demonstrates strong performance in controlled experiments, real world deployment in present challenges. Variations in water turbidity, lighting conditions, and the presence of other marine organisms can affect detection accuracy. Hardware must be used to withstand prolonged submersion, biofouling, and power limitations. To maintaining model accuracy over time will require periodic retraining with updated imagery to account for ecological changes and equipment wear.</p>
</sec>
<sec id="s4_7_2">
<label>4.7.2</label>
<title>Limitation</title>
<p>A potential limitation is overfitting to features present in DCGAN-generated synthetic images, which may not fully represent natural reef complexity. To address these issues, future work will focus on testing the model with newly collected reef imagery from locations and conditions not represented in the CSIRO dataset. Such external testing is essential to ensure the model&#x2019;s robustness across diverse reef environments and to avoid bias toward synthetic data artifacts.</p>
</sec>
<sec id="s4_7_3" sec-type="conclusions">
<label>4.7.3</label>
<title>Conclusion</title>
<p>The proposed work demonstrates that combining focal loss with DCGAN-based synthetic data augmentation can significantly enhance the detection of Crown-of-Thorns Starfish in complex underwater environments. The approach addresses class imbalance, improves feature recognition for underrepresented classes, and broadens the range of training scenarios to strengthen model generalization. Achieving high precision, recall, and mAP scores, the optimized model is well suited for deployment on embedded systems, enabling real time, scalable, and efficient reef monitoring. This framework offers a possible and impactful tool for supporting timely interventions and promoting the long term conservation and ecological resilience of coral reef ecosystems.</p>
</sec>
</sec>
</sec>
</body>
<back>
<sec id="s5" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s6" sec-type="author-contributions">
<title>Author contributions</title>
<p>SP: Methodology, Data curation, Visualization, Project administration, Supervision, Conceptualization, Resources, Investigation, Software, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing, Formal Analysis. JD: Funding acquisition, Formal Analysis, Data curation, Validation, Project administration, Software, Writing &#x2013; review &amp; editing, Writing &#x2013; original draft, Investigation. MJ: Writing &#x2013; original draft, Formal Analysis, Visualization, Methodology, Resources, Investigation, Conceptualization, Validation, Writing &#x2013; review &amp; editing, Software.</p>
</sec>
<sec id="s7" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research and/or publication of this article. The article processing charge will be covered by Vellore Institute of Technology, Chennai.</p>
</sec>
<sec id="s8" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s9" sec-type="ai-statement">
<title>Generative AI statement</title>
<p>The author(s) declare that no Generative AI was used in the creation of this manuscript.</p>
<p>Any alternative text (alt text) provided alongside figures in this article has been generated by Frontiers with the support of artificial intelligence and reasonable efforts have been made to ensure accuracy, including review by the authors wherever possible. If you identify any issues, please contact us.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Er</surname> <given-names>M. J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Dynamic YOLO for small underwater object detection</article-title>. <source>Artif. Intell. Rev.</source> <volume>57</volume>, <fpage>165</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10462-024-10788-1</pub-id>
</citation></ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Chen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Dong</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>WaterPairs: A paired dataset for underwater image enhancement and underwater object detection</article-title>. <source>Intell. Mar. Technol. Syst</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s44295-024-00021-8</pub-id>
</citation></ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cherian</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Venugopal</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Abishek</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Jabbar</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Survey on underwater image enhancement using deep learning</article-title>. <source>Proc. Int. Conf. Comput. Commun. Secur. Intell. Syst. (IC3SIS).</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/IC3SIS54991.2022.9885529</pub-id>
</citation></ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dai</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Ding</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Edge-guided representation learning for underwater object detection</article-title>. <source>CAAI. Trans. Intell. Technol.</source> <volume>9</volume>, <fpage>1078</fpage>&#x2013;<lpage>11091</lpage>.</citation></ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dakhil</surname> <given-names>R. A.</given-names>
</name>
<name>
<surname>Khayeat</surname> <given-names>A. R.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Review on deep learning technique for underwater object detection</article-title>. <source>arXiv. preprint. arXiv:2209.10151</source>.</citation></ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dakhil</surname> <given-names>R. A.</given-names>
</name>
<name>
<surname>Khayeat</surname> <given-names>A. R.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Deep learning for enhanced marine vision: Object detection in underwater environments</article-title>. <source>Int. J. Electr. Electron. Res</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.37391/IJEER</pub-id>
</citation></ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Edge</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Islam</surname> <given-names>M. J.</given-names>
</name>
<name>
<surname>Morse</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Sattar</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A generative approach for detection-driven underwater image enhancement</article-title>. <source>arXiv. preprint. arXiv:2012.05990</source>.</citation></ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fang</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Sheng</surname> <given-names>V. S.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>A method for improving CNN-based image recognition using DCGAN</article-title>. <source>Comput. Mater. Continua.</source> <volume>57</volume>, <fpage>167</fpage>&#x2013;<lpage>1178</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.32604/cmc.2018.02356</pub-id>
</citation></ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fayaz</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Parah</surname> <given-names>S. A.</given-names>
</name>
<name>
<surname>Qureshi</surname> <given-names>G. J.</given-names>
</name>
<name>
<surname>Lloret</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Del Ser</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Muhammad</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Intelligent underwater object detection and image restoration for autonomous underwater vehicles</article-title>. <source>IEEE Trans. Veh. Technol.</source> <volume>73</volume>, <fpage>1726</fpage>&#x2013;<lpage>11735</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TVT.2023.3318629</pub-id>
</citation></ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Feng</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jin</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>CEH-YOLO: A composite enhanced YOLO-based model for underwater object detection</article-title>. <source>Ecol. Inform</source> <volume>82</volume>, <fpage>102758</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2024.102758</pub-id>
</citation></ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Geng</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Bhatti</surname> <given-names>U. A.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>PE-Transformer: Path enhanced transformer for improving underwater object detection</article-title>. <source>Expert Syst. Appl.</source> <volume>246</volume>, <fpage>123253</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2024.123253</pub-id>
</citation></ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Guo</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A lightweight YOLOv8 integrating FasterNet for real-time underwater object detection</article-title>. <source>J. Real-Time. Image. Proc.</source> <volume>21</volume>, <fpage>49</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11554-024-01431-x</pub-id>
</citation></ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Jian</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Tao</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Zhi</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Underwater object detection and datasets: A survey</article-title>. <source>Intell. Mar. Technol. Syst</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s44295-024-00023-6</pub-id>
</citation></ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khriss</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Elmiad</surname> <given-names>A. K.</given-names>
</name>
<name>
<surname>Badaoui</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Barkaoui</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Zarhloule</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Exploring deep learning for underwater plastic debris detection and monitoring</article-title>. <source>J. Ecol. Eng</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.12911/22998993/187970</pub-id>
</citation></ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>UW-DETR: Feature fusion enhanced RT-DETR for improving underwater object detection</article-title>. <source>IEEE Access</source> <volume>12</volume>, <fpage>191967</fpage>&#x2013;<lpage>191979</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2024.3515960</pub-id>
</citation></ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lin</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Underwater object detection method based on learnable query recall mechanism and lightweight adapter</article-title>. <source>PloS One</source> <volume>19</volume>, <elocation-id>e0298739</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1371/journal.pone.0298739</pub-id>, PMID: <pub-id pub-id-type="pmid">38416764</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Peng</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Tang</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Underwater object detection using TC-YOLO with attention mechanisms</article-title>. <source>Sensors. (Basel).</source> <volume>23</volume>,  <elocation-id>2567</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s23052567</pub-id>, PMID: <pub-id pub-id-type="pmid">36904769</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Wen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Huang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Xie</surname> <given-names>N.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Lightweight underwater object detection algorithm for embedded deployment using higher-order information and image enhancement</article-title>. <source>J. Mar. Sci. Eng</source>. <volume>12</volume>, <elocation-id>506</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/jmse12030506</pub-id>
</citation></ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Lv</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Fan</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Luo</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zou</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Application of an improved DCGAN for image generation</article-title>. <source>Mobile. Inf. Syst.</source> <volume>2022</volume>, <fpage>9005552</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1155/2022/9005552</pub-id>
</citation></ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Marine biometric recognition algorithm based on YOLOv3-GAN network</article-title>. <source>Proc. Conf. Multimedia. Modeling</source>.</citation></ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lokanath</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Kumar</surname> <given-names>K. S.</given-names>
</name>
<name>
<surname>Keerthi</surname> <given-names>E. S.</given-names>
</name>
</person-group> (<year>2017</year>). <article-title>Accurate object classification and detection by Faster-RCNN</article-title>. <source>IOP. Conf. Ser.: Mater. Sci. Eng.</source> <volume>263</volume>, <elocation-id>52028</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1088/1757-899X/263/5/052028</pub-id>
</citation></ref>
<ref id="B22">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nambiar</surname> <given-names>T. T. C, A. M.</given-names>
</name>
<name>
<surname>Mittal</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>A GAN-based super resolution model for efficient image enhancement in underwater sonar images</article-title>,&#x201d; in <conf-name>Proc. OCEANS 2022 - Chennai</conf-name>. <fpage>1</fpage>&#x2013;<lpage>8</lpage>.</citation></ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Nguyen</surname> <given-names>Q. T.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Detrimental starfish detection on embedded system: A case study of YOLOv5 deep learning algorithm and TensorFlow Lite framework</article-title>. <source>J. Comput. Sci. Inst.</source> <volume>23</volume>, <fpage>105</fpage>&#x2013;<lpage>1111</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.35784/jcsi.2896</pub-id>
</citation></ref>
<ref id="B24">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Nooka</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Alla</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Bala</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Jyothi</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Venkataraman</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Ramadass</surname> <given-names>G. A.</given-names>
</name>
</person-group> (<year>2022</year>). &#x201c;<article-title>Vision-based deep learning algorithm for underwater object detection and tracking</article-title>,&#x201d; in <conf-name>Proc. OCEANS 2022 - Chennai</conf-name>, IEEE Xplore. <fpage>1</fpage>&#x2013;<lpage>6</lpage>.</citation></ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pagire</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Phadke</surname> <given-names>A. C.</given-names>
</name>
<name>
<surname>Hemant</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A deep learning approach for underwater fish detection</article-title>. <source>J. Integr. Sci. Technol</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.62110/sciencein.jist.2024.v12.765</pub-id>
</citation></ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pavithra</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Cicil Melbin Denny</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>An efficient approach to detect and segment underwater images using Swin Transformer</article-title>. <source>Results. Eng.</source> <volume>23</volume>, <elocation-id>102460</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.rineng.2024.102460</pub-id>. ISSN 2590 - 1230.</citation></ref>
<ref id="B27">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> &#x201c;<article-title>Faster R-CNN: Towards real-time object detection with region proposal networks</article-title>,&#x201d; in <conf-name>Advances in Neural Information Processing Systems (NeurIPS),</conf-name> <volume>28</volume>, <fpage>91</fpage>&#x2013;<lpage>95</lpage>, (<year>2015</year>)., PMID: <pub-id pub-id-type="pmid">27295650</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Shah</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Nabi</surname> <given-names>M. M.</given-names>
</name>
<name>
<surname>Alaba</surname> <given-names>S. Y.</given-names>
</name>
<name>
<surname>Prior</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Caillouet</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Campbell</surname> <given-names>M. D.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). &#x201c;<article-title>A zero shot detection based approach for fish species recognition in underwater environments</article-title>,&#x201d; in <conf-name>Proc. OCEANS 2023 - MTS/IEEE U.S. Gulf Coast</conf-name>. <fpage>1</fpage>&#x2013;<lpage>7</lpage>.</citation></ref>
<ref id="B29">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Singhal</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sharma</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Tiwari</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Astya</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Kushwaha</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2025</year>). <article-title>Cognitive analysis of underwater object detection using deep learning for marine exploration</article-title>. <source>Proc. Int. Conf. Eng. Technol. Manage. (ICETM).</source>, <fpage>1</fpage>&#x2013;<lpage>6</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ICETM63734.2025.11051574</pub-id>
</citation></ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Walia</surname> <given-names>J. S.</given-names>
</name>
<name>
<surname>Haridass</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Kumaresan</surname> <given-names>P. L.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Deep learning innovations for underwater waste detection: An in-depth analysis</article-title>. <source>IEEE Access</source> <volume>13</volume>, <fpage>88917</fpage>&#x2013;<lpage>888929</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2025.3569344</pub-id>
</citation></ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Cong</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Fang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Kwong</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>CA-GAN: Class-condition attention GAN for underwater image enhancement</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>130719</fpage>&#x2013;<lpage>130728</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/Access.6287639</pub-id>
</citation></ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Zhong</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Simultaneous restoration and super-r esolution GAN for underwater image enhancement</article-title>. <source>Front. Mar. Sci</source>. <volume>10</volume>, <elocation-id>1162295</elocation-id>.</citation></ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Ren</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Is underwater image enhancement all object detectors need</article-title>? <source>IEEE J. Ocean. Eng.</source> <volume>49</volume>, <fpage>606</fpage>&#x2013;<lpage>6621</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JOE.2023.3302888</pub-id>
</citation></ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Xiao</surname> <given-names>N.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Underwater object detection method based on improved Faster RCNN</article-title>. <source>Appl. Sci.</source> <volume>13</volume>, <elocation-id>2746</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/app13042746</pub-id>
</citation></ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wu</surname> <given-names>Q.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Meng</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>DCGAN-based data augmentation for tomato leaf disease identification</article-title>. <source>IEEE Access</source> <volume>8</volume>, <fpage>98716</fpage>&#x2013;<lpage>998728</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/ACCESS.2020.2997001</pub-id>
</citation></ref>
<ref id="B36">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Long</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>Z.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>UCDN: A CenterNet-based dense multi-scale detection fusion net on underwater objects,&#x201d; in Proc</article-title>. <source>IEEE Int. Conf. Comput. Commun. Artif. Intell. (CCAI).</source> <volume>pp</volume>, <fpage>249</fpage>&#x2013;<lpage>254</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/CCAI57533.2023.10201320</pub-id>
</citation></ref>
<ref id="B37">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>An improved YOLOv5-based underwater object-detection framework</article-title>. <source>Sensors. (Basel).</source> <volume>23</volume>, <fpage>8287</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/s23073693</pub-id>, PMID: <pub-id pub-id-type="pmid">37050753</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Cao</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Gao</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Song</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Underwater object detection algorithm based on an improved YOLOv8</article-title>. <source>J. Mar. Sci. Eng</source>. <volume>12</volume>, <elocation-id>1991</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/jmse12111991</pub-id>
</citation></ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhou</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>BG-YOLO: A bidirectional-guided method for underwater object detection</article-title>. <source>Sensors. (Basel).</source> <volume>24</volume>., PMID: <pub-id pub-id-type="pmid">39599187</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Zhu</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Gan</surname> <given-names>W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>YOLOv7t-CEBC network for underwater litter detection</article-title>. <source>J. Mar. Sci. Eng</source>. <volume>23</volume>, <fpage>8287</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/jmse12040524</pub-id>
</citation></ref>
<ref id="B41">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Letcher</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Fair</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Bringing vision to climate: A hierarchical model for water depth monitoring in headwater streams</article-title>. <source>Inf. Fusion.</source> <volume>110</volume>, <fpage>102448</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.inffus.2024.102448</pub-id>
</citation></ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Kong</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Yuan</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Pan</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>R.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>Real-time underwater object detection technology for complex underwater environments based on deep learning</article-title>. <source>Ecol. Inform</source> <volume>82</volume>, <fpage>102680</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2024.102680</pub-id>
</citation></ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhou</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Qin</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Dai</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Sonar image generation by MFA-CycleGAN for boosting underwater object detection of AUVs</article-title>. <source>IEEE J. Ocean. Eng.</source> <volume>49</volume>, <fpage>905</fpage>&#x2013;<lpage>9919</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/JOE.2024.3350746</pub-id>
</citation></ref>
</ref-list>
</back>
</article>