<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurorobot.</journal-id>
<journal-title>Frontiers in Neurorobotics</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurorobot.</abbrev-journal-title>
<issn pub-type="epub">1662-5218</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbot.2024.1375886</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Real-time precision detection algorithm for jellyfish stings in neural computing, featuring adaptive deep learning enhanced by an advanced YOLOv4 framework</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Zhu</surname> <given-names>Chao</given-names></name>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2639298/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Feng</surname> <given-names>Hua</given-names></name>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Xu</surname> <given-names>Liang</given-names></name>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff><institution>Emergency Department of Qinhuangdao First Hospital, Qinhuangdao</institution>, <addr-line>Hebei</addr-line>, <country>China</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Xianmin Wang, Guangzhou University, China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Jian Liu, Hunan University, China</p>
<p>Noran Ouf, Cairo University, Egypt</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Chao Zhu <email>18503388058&#x00040;163.com</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>23</day>
<month>05</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1375886</elocation-id>
<history>
<date date-type="received">
<day>24</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>29</day>
<month>04</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2024 Zhu, Feng and Xu.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Zhu, Feng and Xu</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Sea jellyfish stings pose a threat to human health, and traditional detection methods face challenges in terms of accuracy and real-time capabilities.</p></sec>
<sec>
<title>Methods</title>
<p>To address this, we propose a novel algorithm that integrates YOLOv4 object detection, an attention mechanism, and PID control. We enhance YOLOv4 to improve the accuracy and real-time performance of detection. Additionally, we introduce an attention mechanism to automatically focus on critical areas of sea jellyfish stings, enhancing detection precision. Ultimately, utilizing the PID control algorithm, we achieve adaptive adjustments in the robot&#x00027;s movements and posture based on the detection results. Extensive experimental evaluations using a real sea jellyfish sting image dataset demonstrate significant improvements in accuracy and real-time performance using our proposed algorithm. Compared to traditional methods, our algorithm more accurately detects sea jellyfish stings and dynamically adjusts the robot&#x00027;s actions in real-time, maximizing protection for human health.</p></sec>
<sec>
<title>Results and discussion</title>
<p>The significance of this research lies in providing an efficient and accurate sea jellyfish sting detection algorithm for intelligent robot systems. The algorithm exhibits notable improvements in real-time capabilities and precision, aiding robot systems in better identifying and addressing sea jellyfish stings, thereby safeguarding human health. Moreover, the algorithm possesses a certain level of generality and can be applied to other applications in target detection and adaptive control, offering broad prospects for diverse applications.</p></sec></abstract>
<kwd-group>
<kwd>YOLOv4</kwd>
<kwd>attention mechanism</kwd>
<kwd>PID</kwd>
<kwd>jellyfish stings</kwd>
<kwd>intelligent robotics</kwd>
<kwd>computer vision</kwd>
</kwd-group>
<counts>
<fig-count count="11"/>
<table-count count="8"/>
<equation-count count="16"/>
<ref-count count="32"/>
<page-count count="17"/>
<word-count count="9003"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1 Introduction</title>
<p>The development of intelligent robots has been widely applied across various fields, including target detection and adaptive control (Martin-Abadal et al., <xref ref-type="bibr" rid="B16">2020</xref>). In tasks such as marine exploration and rescue missions, detecting sea Jellyfish stings is crucial due to the threat they pose to human health. However, traditional detection methods face challenges in terms of accuracy and real-time capabilities, necessitating the development of a new algorithm (Cunha and Dinis-Oliveira, <xref ref-type="bibr" rid="B3">2022</xref>). The purpose of this paper is to propose an adaptive intelligent robot algorithm for real-time and accurate sea Jellyfish sting detection, based on an improved Yolov4, attention mechanism, and PID control. This algorithm aims to enhance the accuracy and real-time performance of sea Jellyfish sting detection, thereby better safeguarding human health (Cunha and Dinis-Oliveira, <xref ref-type="bibr" rid="B3">2022</xref>). Here are five commonly used deep learning or machine learning models in the fields of target detection and adaptive control:</p>
<p>YOLO (You Only Look Once) (Gao M. et al., <xref ref-type="bibr" rid="B6">2021</xref>) is a fast and real-time object detection model. It employs a single neural network to perform object detection in a single forward pass, making it suitable for applications with high real-time requirements. YOLO&#x00027;s network structure is relatively simple, and both training and inference processes are efficient. By dividing the image into a grid, with each grid predicting the bounding box and category of the target, YOLO can capture global contextual information. However, YOLO exhibits lower detection accuracy for small and dense targets, and its localization precision is limited.</p>
<p>Faster R-CNN (Region-based Convolutional Neural Network) (Zeng et al., <xref ref-type="bibr" rid="B30">2021</xref>) is an object detection model with high detection accuracy. It achieves object detection through two main steps: extracting candidate regions and classifying and locating these regions. Faster R-CNN excels in detection accuracy and can handle various target sizes and densities. However, due to the need for multiple steps and complex computations, Faster R-CNN has a relatively slower speed and is not suitable for real-time applications.</p>
<p>SSD (Single Shot MultiBox Detector) (Ma et al., <xref ref-type="bibr" rid="B14">2021</xref>) is a fast object detection model suitable for real-time applications. SSD detects targets by applying a convolutional sliding window on feature maps of different scales. It has good detection speed and high accuracy, adapting well to targets of different sizes. However, compared to other models, SSD&#x00027;s detection accuracy for small targets is relatively lower.</p>
<p>RetinaNet (Liu et al., <xref ref-type="bibr" rid="B13">2023</xref>) is an object detection model that performs well in handling small targets. It introduces a novel loss function that balances samples with different target sizes. It exhibits good performance in detecting small targets, effectively addressing the issue of small targets being easily overlooked. However, its detection accuracy is relatively lower when dealing with dense and large targets.</p>
<p>Mask R-CNN (Nie et al., <xref ref-type="bibr" rid="B17">2020</xref>) is an object detection model capable of pixel-level segmentation of targets. In addition to detecting the bounding box and category of targets, Mask R-CNN can generate precise masks for targets. This makes Mask R-CNN highly useful when detailed target segmentation information is required. However, due to the need for pixel-level predictions, Mask R-CNN has a relatively slower speed.</p>
<p>The following are three related research directions:</p>
<p>Improving small object detection accuracy in real-time object detection models. Real-time object detection plays a crucial role in various application domains, but current real-time models face challenges in achieving high accuracy for small object detection (Mahaur et al., <xref ref-type="bibr" rid="B15">2023</xref>). To enhance the small object detection accuracy in real-time object detection models, research can focus on the following aspects: Firstly, improving feature representation capabilities. Secondly, designing more refined object detection loss functions. Existing object detection loss functions may have issues with small objects as they tend to prioritize larger targets (Khamassi et al., <xref ref-type="bibr" rid="B10">2023</xref>). By researching and improving in the above directions, the performance of real-time object detection models in small object detection accuracy can be enhanced, expanding their applicability to a wider range of real-world scenarios (Zhang et al., <xref ref-type="bibr" rid="B31">2022</xref>).</p>
<p>Integrating multimodal information in object detection models (Chen et al., <xref ref-type="bibr" rid="B2">2019</xref>). Object detection is typically based on image data, but in some application scenarios, combining multimodal information from other sensors may provide more accurate and comprehensive object detection results (Gao W. et al., <xref ref-type="bibr" rid="B7">2021</xref>). Therefore, researching object detection models that integrate multimodal information is a promising direction. One approach is to fuse image data with other sensor data to improve detection accuracy and robustness (Wu et al., <xref ref-type="bibr" rid="B26">2021</xref>).</p>
<p>Designing and optimizing lightweight object detection models. In resource-constrained scenarios like embedded devices or mobile platforms, there is a demand for object detection models with small model sizes and low computational complexity while maintaining high detection accuracy (Han et al., <xref ref-type="bibr" rid="B8">2022</xref>). Therefore, designing and optimizing lightweight object detection models is a challenging and practical direction (Li et al., <xref ref-type="bibr" rid="B11">2018</xref>). One approach is to reduce model size and computational complexity through network compression and model pruning (Huang et al., <xref ref-type="bibr" rid="B9">2018</xref>). Exploring the use of lightweight network structures such as MobileNet and ShuffleNet for fine-tuning on object detection tasks is one option. Additionally, techniques like parameter sharing, channel pruning, and quantization can reduce model parameters and computations for designing lightweight object detection models (Lin and Xu, <xref ref-type="bibr" rid="B12">2023</xref>).</p>
<p>Traditional sea Jellyfish sting detection methods face issues in accuracy and real-time capabilities. Therefore, we propose a new algorithm that integrates improved Yolov4, attention mechanism, and PID control to enhance detection accuracy and real-time performance. Firstly, we enhance Yolov4 to improve the accuracy and real-time performance of detection. This involves adjusting network architecture, loss functions, and data augmentation strategies to adapt Yolov4 for sea Jellyfish sting detection tasks. Secondly, we introduce an attention mechanism to automatically focus on critical areas of sea Jellyfish stings, enhancing detection precision. Using attention mechanisms such as SENet or SAM enhances the model&#x00027;s focus on target areas, improving accuracy and robustness. Lastly, we employ the PID control algorithm to achieve adaptive adjustments in the robot&#x00027;s movements and posture based on detection results. The PID control algorithm adjusts parameters in response to error signals, enabling real-time and precise control based on detected sea Jellyfish stings. In the field of sea Jellyfish sting detection, traditional methods face challenges in accuracy and real-time capabilities. Thus, we propose an adaptive intelligent robot algorithm for real-time and accurate sea Jellyfish sting detection, integrating improved Yolov4, attention mechanism, and PID control. This algorithm addresses issues with traditional methods and enhances the ability to protect human health.</p>
<list list-type="bullet">
<list-item><p>Comprehensive comparison of different object detection models: This paper provides a comprehensive comparison of five commonly used object detection models, namely YOLO, Faster R-CNN, SSD, RetinaNet, and Mask R-CNN. By analyzing their strengths and weaknesses, readers can gain a better understanding of each model&#x00027;s characteristics, enabling them to choose the most suitable model for their specific application scenarios.</p></list-item>
<list-item><p>Emphasis on model applicability and limitations: The paper underscores the applicability and limitations of each model. This information assists readers in selecting the most appropriate object detection model based on their individual needs and application contexts. For instance, if real-time performance is a priority, faster models like YOLO or SSD may be preferred. Conversely, if higher detection accuracy is required, Faster R-CNN or RetinaNet might be more suitable.</p></list-item>
<list-item><p>Providing a comprehensive understanding of object detection models: The paper offers brief introductions to the principles and features of each model, enabling readers to gain a comprehensive understanding of object detection models. This knowledge empowers readers to delve deeper into the research and application of object detection technology, making informed decisions in practical projects.</p></list-item>
</list></sec>
<sec sec-type="methods" id="s2">
<title>2 Methodology</title>
<sec>
<title>2.1 Overview of our network</title>
<p>The Adaptive Intelligent Robot Real-time Accurate Detection Algorithm for Sea Jellyfish Sting Injuries, based on Improved YOLOv4 and Attention Mechanism combined with PID Control, aims to achieve precise detection and identification of sting injuries in the marine environment. This is accomplished by integrating object detection, attention mechanism, and control algorithms to adaptively adjust the robot&#x00027;s actions in response to changes and errors during the detection process. <xref ref-type="fig" rid="F1">Figure 1</xref> represents the overall schematic diagram of the proposed model.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>The overall schematic diagram of the proposed model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0001.tif"/>
</fig>


<p>Optimize the network structure, training strategies, and loss functions of YOLOv4 to enhance the accuracy and efficiency of the object detection algorithm. Introduce an attention mechanism to enable the algorithm to focus on important image regions, improving detection accuracy and robustness. This can be achieved by adding attention modules to the network or by adjusting feature map weights. Design a PID control algorithm to utilize the error between detection results and expected values to adjust the robot&#x00027;s actions and behaviors. This adaptation is crucial to cope with variations and errors encountered during the detection process.</p>
<p>Overall implementation process:</p>
<list list-type="bullet">
<list-item><p>Data collection and preparation: Gather images or video data from the marine environment and preprocess it, including tasks such as image enhancement and noise reduction.</p></list-item>
<list-item><p>Design of object detection network: Design and enhance the YOLOv4 network structure, involving adjustments to network layers, the introduction of new feature extraction modules, or optimization of loss functions.</p></list-item>
<list-item><p>Introduction of attention mechanism: Incorporate an attention mechanism into the object detection network, allowing the model to concentrate on crucial image regions. This can be achieved by adding attention modules or adjusting feature map weights within the network.</p></list-item>
<list-item><p>Design of PID control algorithm: Develop a PID control algorithm to dynamically adjust the robot&#x00027;s actions and behaviors based on the error between detection results and expected values. The PID algorithm encompasses proportional, integral, and derivative control parameters.</p></list-item>
<list-item><p>Training and optimization: Train the improved network using annotated data and optimize network parameters and attention mechanisms through the iterative process of backpropagation. This optimization is performed iteratively on training and validation sets. Real-time Detection and Feedback:</p></list-item>
</list>
<p>Deploy the trained model and control algorithm to the intelligent robot for real-time detection and feedback in the marine environment. The robot captures marine images or videos, feeds them into the object detection network for real-time sting injury detection, and adjusts its actions based on the comparison between detection results and expected values. This adaptation allows the robot to accommodate changes and errors encountered during the detection process.</p></sec>
<sec>
<title>2.2 Advanced YOLOv4 model</title>
<p>Advanced YOLOv4 is an improved version of the traditional YOLOv4 object detection algorithm, designed to enhance detection accuracy and efficiency. The following details the fundamental principles and roles of the Advanced YOLOv4 model in this approach (Roy et al., <xref ref-type="bibr" rid="B19">2022</xref>). Advanced YOLOv4 incorporates a series of improvements, including adjustments to the network structure, optimization of feature extraction modules, enhancement of loss functions, and optimization of training strategies. These improvements aim to enhance the performance and speed of the object detection algorithm (Wang and Liu, <xref ref-type="bibr" rid="B25">2022</xref>). <xref ref-type="fig" rid="F2">Figure 2</xref> shows the schematic diagram of the proposed Advanced YOLOv4 model.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>The schematic diagram of the proposed Advanced YOLOv4 model.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0002.tif"/>
</fig>


<p>Network structure adjustments:</p>
<p>Advanced YOLOv4 modifies the YOLOv4 network structure by introducing additional convolutional layers and residual connections, thereby enhancing the network&#x00027;s representational and feature extraction capabilities. Optimization of Feature Extraction Modules:</p>
<p>The model adopts CSPDarknet53 as the primary feature extraction module, combining Cross-Stage Partial connections and the structure of Darknet53. This integration better extracts image features, contributing to improved detection accuracy. Improved Loss Function:</p>
<p>Advanced YOLOv4 utilizes an enhanced loss function known as the Generalized Intersection over Union (GIoU) loss function. This function considers the overlap of target boxes when calculating position and size errors, providing a more accurate measure of target box matching. Optimized Training Strategy:</p>
<p>The model employs a multi-scale training strategy, training the model on images at different scales. This approach enhances the model&#x00027;s adaptability to targets of varying sizes.</p>
<p>&#x0201C;GhostNet&#x0201D; is a lightweight convolutional neural network architecture that introduces &#x0201C;ghost&#x0201D; modules, which use fewer parameters and computational resources in each convolutional layer, thereby achieving higher computational efficiency. In the Advanced YOLOv4 model, we have incorporated &#x0201C;GhostNet&#x0201D; as part of the base network structure to enhance the model&#x00027;s lightweight characteristics, speed up the detection process, and reduce the computational resource requirements of the model.</p>
<p>&#x0201C;Depthwise Separable Convolution&#x0201D; is a type of convolution operation that decomposes standard convolution into two steps: depthwise convolution and pointwise convolution. This decomposition significantly reduces the number of parameters and computational load in the model, thereby improving the model&#x00027;s computational efficiency and speed. In the Advanced YOLOv4 model, we have adopted &#x0201C;Depthwise Separable Convolution&#x0201D; as part of the convolution operations to accelerate the model&#x00027;s inference process and enable faster real-time detection.</p>
<p>Role in the Method: Advanced YOLOv4 plays a crucial role in the Adaptive Intelligent Robot Real-time Accurate Detection Algorithm for Sea Jellyfish Sting Injuries, which combines improved YOLOv4 and attention mechanisms with PID control.</p>
<list list-type="bullet">
<list-item><p>Improved detection accuracy: Through network structure adjustments and feature extraction module optimization, Advanced YOLOv4 better extracts image features, thereby enhancing the accuracy of object detection. This is crucial for precise detection and identification of sea Jellyfish sting injuries.</p>
<p>Enhanced detection efficiency: Optimization of the network structure and training strategies in Advanced YOLOv4 contributes to improved speed and efficiency of the object detection algorithm. This is crucial for real-time detection and feedback, enabling intelligent robots to respond promptly to detection results.</p>
</list-item>
<list-item><p>Improved loss function impact: The use of the GIoU loss function in Advanced YOLOv4 contributes to more accurate measurement of target box matching. This aids in improving detection precision and provides more accurate error signals for adaptive control.</p></list-item>
</list>
<p>The formula for Advanced YOLOv4 is as follows (<xref ref-type="disp-formula" rid="E1">Equation 1</xref>):</p>
<p>Coordinate loss term:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable columnalign='left'><mml:mtr><mml:mtd><mml:mtext>coord</mml:mtext><mml:mo>&#x0005F;</mml:mo><mml:mtext>loss&#x02009;&#x02009;=&#x02009;</mml:mtext><mml:msub><mml:mi>&#x003BB;</mml:mi><mml:mrow><mml:mtext>coord</mml:mtext></mml:mrow></mml:msub><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mi>S</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mi>B</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mo>&#x022AE;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mtext>obj</mml:mtext></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mover accent='true'><mml:mi>x</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mtext>2</mml:mtext></mml:msup><mml:mtext>+</mml:mtext><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>y</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mover accent='true'><mml:mi>y</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mtext>2</mml:mtext></mml:msup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mtext>&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;&#x02009;+&#x02009;</mml:mtext><mml:msub><mml:mi>&#x003BB;</mml:mi><mml:mrow><mml:mtext>coord</mml:mtext></mml:mrow></mml:msub><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mi>S</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mtext>j=0</mml:mtext></mml:mrow><mml:mi>B</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mo>&#x022AE;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mtext>obj</mml:mtext></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msqrt><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msqrt><mml:msqrt><mml:mrow><mml:msub><mml:mover accent='true'><mml:mi>w</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msqrt><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mtext>2</mml:mtext></mml:msup><mml:mtext>+</mml:mtext><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msqrt><mml:mrow><mml:msub><mml:mi>h</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msqrt><mml:msqrt><mml:mrow><mml:msub><mml:mover accent='true'><mml:mi>h</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:msqrt><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mtext>2</mml:mtext></mml:msup></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Among them, &#x003BB;<sub>coord</sub> is the weight parameter of the coordinate loss, <italic>S</italic> is the size of the feature map, <italic>B</italic> is the number of bounding boxes predicted for each grid, <inline-formula><mml:math id="M2"><mml:msubsup><mml:mrow><mml:mo>&#x022AE;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mstyle class="text"><mml:mtext class="textrm" mathvariant="normal">obj</mml:mtext></mml:mstyle></mml:mrow></mml:msubsup></mml:math></inline-formula> represents the indicator function of whether the <italic>j</italic>-th bounding box in the <italic>i</italic>-th grid contains the target, <italic>x</italic><sub><italic>i</italic></sub>, <italic>y</italic><sub><italic>i</italic></sub> is the <italic>j</italic>-th boundary box in the <italic>i</italic>-th grid The center coordinates of the bounding box, <inline-formula><mml:math id="M3"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>&#x00177;</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are the predicted center coordinates of the <italic>j</italic>-th bounding box in the <italic>i</italic>-th grid, <italic>w</italic><sub><italic>i</italic></sub>, <italic>h</italic><sub><italic>i</italic></sub> are the &#x02212;<italic>thThewidthandheightofthe</italic>j&#x02212;<italic>thboundingboxinthei</italic>-th grid, &#x00175;<sub><italic>i</italic></sub>, &#x00125;<sub><italic>i</italic></sub> are the predicted widths of the <italic>j</italic>-th bounding box in the <italic>i</italic>-th grid and height (<xref ref-type="disp-formula" rid="E2">Equation 2</xref>).</p>
<p>Category loss items:</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M4"><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:msub><mml:mi>d</mml:mi><mml:mi>l</mml:mi></mml:msub><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mtext>=</mml:mtext><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:msup><mml:mi>S</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:munderover><mml:mrow><mml:mstyle displaystyle='true'><mml:munderover><mml:mo>&#x02211;</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:mrow><mml:mi>B</mml:mi></mml:munderover><mml:mrow><mml:msubsup><mml:mo>&#x022AE;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow><mml:mrow><mml:mtext>obj</mml:mtext></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:mrow></mml:mstyle><mml:msup><mml:mrow><mml:mo stretchy='false'>(</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x02212;</mml:mo><mml:msub><mml:mover accent='true'><mml:mi>C</mml:mi><mml:mo>&#x0005E;</mml:mo></mml:mover><mml:mi>i</mml:mi></mml:msub><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mtext>2</mml:mtext></mml:msup></mml:mrow></mml:math></disp-formula>
<p>Among them, <italic>C</italic><sub><italic>i</italic></sub> is the category confidence score of the <italic>j</italic>-th bounding box in the <italic>i</italic>-th grid, and &#x00108;<sub><italic>i</italic></sub> is the <italic>j</italic>-th bounding box in the <italic>i</italic>-th grid (<xref ref-type="disp-formula" rid="E2">Equation 2</xref>). Predicted class confidence score.</p>
<p>The final loss function is:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M5"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mstyle mathvariant="bold"><mml:mtext>L</mml:mtext></mml:mstyle><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">&#x000A0;</mml:mtext><mml:mstyle class="math"><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mtext class="textrm" mathvariant="normal">&#x000A0;</mml:mtext></mml:mstyle><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textrm" mathvariant="normal">&#x000A0;</mml:mtext><mml:mstyle class="math"><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:msub><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mtext class="textrm" mathvariant="normal">&#x000A0;</mml:mtext></mml:mstyle><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textrm" mathvariant="normal">&#x000A0;</mml:mtext><mml:mstyle class="math"><mml:mi>o</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mi>e</mml:mi><mml:msub><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mi>o</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mtext class="textrm" mathvariant="normal">&#x000A0;</mml:mtext></mml:mstyle></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>This loss function will be optimized during training to minimize the difference between the predicted and ground-truth boxes (<xref ref-type="disp-formula" rid="E3">Equation 3</xref>). By adjusting the weight parameters and optimization algorithm, the performance of the target detection model can be improved.</p>
<p>This formula describes the loss function of Advanced YOLOv4, which includes coordinate loss terms and category loss terms. The coordinate loss term measures the difference between the location and size predictions of the object&#x00027;s bounding box and the ground truth, while the category loss term measures the difference between the class confidence prediction of the object and the ground truth.</p>
<p>In summary, Advanced YOLOv4, through enhancements in network structure, feature extraction modules, loss functions, and training strategies, elevates the performance and speed of the object detection algorithm. It plays a key role in the Adaptive Intelligent Robot Real-time Accurate Detection Algorithm for Sea Jellyfish Sting Injuries, based on improved YOLOv4 and attention mechanisms combined with PID control.</p></sec>
<sec>
<title>2.3 Attention mechanism</title>
<p>Attention Mechanism is a method that simulates human visual or auditory attention and is widely used in deep learning models, especially in Natural Language Processing (NLP) and Computer Vision (CV) tasks (Obeso et al., <xref ref-type="bibr" rid="B18">2022</xref>). The fundamental idea of the attention mechanism is that, given an input sequence and a query (or key information), the model calculates the degree of correlation between each input position and the query. It assigns a weight to each input position, representing the model&#x00027;s focus or importance for different input positions. Then, by taking the weighted sum of the features at input positions using their corresponding weights, the final context representation is obtained (Gao et al., <xref ref-type="bibr" rid="B5">2020</xref>). In NLP tasks, the input sequence can be a sentence or a text sequence, and the query can be a specific word or position. In CV tasks, the input sequence can be the feature map of an image, and the query can be a spatial position or a specific region of the image.</p>
<p><xref ref-type="fig" rid="F3">Figure 3</xref> shows the schematic diagram of the Attention Mechanism.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>The schematic diagram of the Attention Mechanism.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0003.tif"/>
</fig>


<p>In attention mechanisms, the most commonly used is soft attention, and its computation process is as follows:</p>
<p>Calculation of correlation between the input sequence and the query: this is done by computing similarity scores between each position in the input sequence and the query, using methods like dot product, scaled dot product, bilinear, or multi-layer perceptron.</p>
<p>Normalization of correlation: to obtain the weight for each position, normalization of the correlation is performed. The softmax function is often used to convert scores into a probability distribution, ensuring that the weights sum up to 1.</p>
<p>Calculation of context representation: the final context representation is obtained by taking the weighted sum of the features in the input sequence using the normalized weights. This context representation can be used for subsequent computations or tasks.</p>
<p>Functions: Attention mechanisms play a crucial role in deep learning models, offering several advantages:</p>
<p>Focus on important information: By calculating the weight for each position, the model can automatically focus on relevant and crucial information in the input sequence related to the query. This enables the model to handle long sequences or large inputs more effectively and extract key features relevant to the task.</p>
<p>Context awareness: Attention mechanisms allow the model to consider information from other positions while processing each position. This context awareness helps improve the model&#x00027;s understanding and generalization capabilities.</p>
<p>Flexibility and interpretability: Attention mechanisms are flexible and can be designed and adjusted according to the requirements of the task. Additionally, the distribution of attention weights provides interpretability, allowing us to understand which parts of the input the model is focusing on.</p>
<p>The formula of the attention mechanism is as follows:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">Attention</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>Q</mml:mi><mml:mo>,</mml:mo><mml:mi>K</mml:mi><mml:mo>,</mml:mo><mml:mi>V</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mtext class="textrm" mathvariant="normal">softmax</mml:mtext><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:mi>Q</mml:mi><mml:msup><mml:mrow><mml:mi>K</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow><mml:mi>V</mml:mi></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Among them, the explanation of variables is as follows <xref ref-type="disp-formula" rid="E4">Equation (4)</xref>:</p>
<p><italic>Q</italic>: query matrix, indicating the location or information that the model focuses on. <italic>K</italic>: key matrix, representing the position or feature of the input sequence. <italic>V</italic>: value matrix, representing the characteristics of the input sequence. <italic>d</italic><sub><italic>k</italic></sub>: dimension of the key matrix (usually the number of columns of the key matrix). softmax: softmax function, used to convert scores into probability distributions. <italic>T</italic>: Transpose operation, transpose the matrix. The calculation process of the attention mechanism is to do the dot product of the query matrix and the key matrix, then divide it by a scaling factor <inline-formula><mml:math id="M7"><mml:msqrt><mml:mrow><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msqrt></mml:math></inline-formula>, and finally obtain the weight through the softmax function. These weights are weighted and summed with the value matrix to obtain the final context representation.</p>
<p>Attention mechanisms enable models to dynamically and selectively focus on different parts of a sequence when processing sequential data, thereby enhancing the model&#x00027;s performance and capabilities. It has achieved significant success in various NLP and CV tasks and remains a hot topic in current deep learning research.</p></sec>
<sec>
<title>2.4 PID algorithm</title>
<p>The PID algorithm (Proportional-Integral-Derivative) (Vuong and Nguyen, <xref ref-type="bibr" rid="B24">2023</xref>) is a classical control algorithm used for implementing adaptive control systems. The PID algorithm adjusts the controller&#x00027;s output based on the current error, past accumulated error, and rate of change of the error to achieve the desired adjustment of the system&#x00027;s dynamic characteristics (Xu et al., <xref ref-type="bibr" rid="B27">2023</xref>). <xref ref-type="fig" rid="F4">Figure 4</xref> shows the schematic diagram of the PID algorithm.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>The schematic diagram of the PID algorithm.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0004.tif"/>
</fig>


<p>The basic principle of the PID algorithm is to continuously adjust the controller&#x00027;s output to minimize the error between the actual system output and the desired output. It consists of three main control components:</p>
<p>Proportional term: The proportional term is directly proportional to the current error and generates a control output proportional to the error magnitude. The proportional term provides a fast response to system changes but may result in steady-state error.</p>
<p>Integral term: The integral term is proportional to the accumulated past errors and is used to handle steady-state errors in the system. The integral term helps eliminate steady-state errors but may lead to overresponse or oscillations.</p>
<p>Derivative term: The derivative term is proportional to the rate of change of the error and is used to predict the future trend of the system. The derivative term helps dampen oscillations and provide a fast response, but it may also result in excessive sensitivity.</p>
<p>The PID algorithm calculates the controller&#x00027;s output by weighted summation of the system&#x00027;s actual error, rate of change of the error, and accumulated error. The formula for the PID algorithm is as follows:</p>
<disp-formula id="E5"><label>(5)</label><mml:math id="M8"><mml:mrow><mml:mi>u</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>=</mml:mo><mml:msub><mml:mi>K</mml:mi><mml:mi>p</mml:mi></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mi>e</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mo>+</mml:mo><mml:msub><mml:mi>K</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mstyle displaystyle='true'><mml:mrow><mml:msubsup><mml:mo>&#x0222B;</mml:mo><mml:mn>0</mml:mn><mml:mi>t</mml:mi></mml:msubsup><mml:mi>e</mml:mi></mml:mrow></mml:mstyle><mml:mo stretchy='false'>(</mml:mo><mml:mi>&#x003C4;</mml:mi><mml:mo stretchy='false'>)</mml:mo><mml:mi>d</mml:mi><mml:mi>&#x003C4;</mml:mi><mml:mo>+</mml:mo><mml:msub><mml:mi>K</mml:mi><mml:mi>d</mml:mi></mml:msub><mml:mo>&#x000B7;</mml:mo><mml:mfrac><mml:mrow><mml:mi>d</mml:mi><mml:mi>e</mml:mi><mml:mo stretchy='false'>(</mml:mo><mml:mi>t</mml:mi><mml:mo stretchy='false'>)</mml:mo></mml:mrow><mml:mrow><mml:mi>d</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula>
<p>where <xref ref-type="disp-formula" rid="E5">Equation (5)</xref>:</p>
<p><italic>u</italic>(<italic>t</italic>) is the controller&#x00027;s output at time <italic>t</italic>. <italic>e</italic>(<italic>t</italic>) is the error of the system, defined as the difference between the desired output and the actual output. <italic>K</italic><sub><italic>p</italic></sub> is the gain coefficient for the proportional term, which adjusts the influence of the proportional control. <italic>K</italic><sub><italic>i</italic></sub> is the gain coefficient for the integral term, which adjusts the influence of the integral control. <italic>K</italic><sub><italic>d</italic></sub> is the gain coefficient for the derivative term, which adjusts the influence of the derivative control.</p>
<p>The PID algorithm aims to continuously adjust the controller&#x00027;s output to gradually approach the desired output and maintain it near the setpoint. By properly setting the PID parameters, the system&#x00027;s stability, fast response, and accurate control can be achieved.</p></sec></sec>
<sec id="s3">
<title>3 Experiment</title>
<sec>
<title>3.1 Datasets</title>
<p>In this paper, we conduct experiments using four datasets.</p>
<p>COCO dataset (common objects in context): The COCO dataset Sharma (<xref ref-type="bibr" rid="B20">2021</xref>) is a widely used large-scale dataset for object detection, segmentation, and captioning tasks. It consists of a diverse collection of images with over 80 common object categories, captured in various contexts. The dataset provides bounding box annotations for object detection, pixel-level segmentations for semantic segmentation, and captions for image captioning. COCO is popular among researchers and used as a benchmark for evaluating object detection and segmentation algorithms.</p>
<p>Pascal VOC dataset (visual object classes): The Pascal VOC dataset Tong and Wu (<xref ref-type="bibr" rid="B22">2023</xref>) is another widely used dataset for object detection, segmentation, and classification tasks. It was created for the annual Visual Object Classes challenge and consists of images from 20 different object categories, including animals, vehicles, and common objects. The dataset provides bounding box annotations for object detection, segmentation masks for semantic segmentation, and class labels for classification. Pascal VOC has been widely used for evaluating and comparing various computer vision algorithms.</p>
<p>KITTI dataset: The KITTI dataset Al-refai and Al-refai (<xref ref-type="bibr" rid="B1">2020</xref>) is specifically designed for autonomous driving and computer vision tasks related to self-driving cars. It includes various sensor modalities such as stereo cameras, LIDAR, and GPS/IMU data. The dataset contains a large number of annotated images captured from a car-mounted sensor suite, covering scenes from urban environments. It provides annotations for tasks such as object detection, tracking, road segmentation, and depth estimation. The KITTI dataset is commonly used for developing and evaluating algorithms related to autonomous driving and scene understanding.</p>
<p>Open Images dataset: The Open Images dataset Veit et al. (<xref ref-type="bibr" rid="B23">2017</xref>) is a large-scale dataset that aims to provide diverse and comprehensive visual data for various computer vision tasks. It contains millions of images from a wide range of categories, covering objects, scenes, and activities. The dataset provides annotations for object detection, segmentation, and visual relationship detection. Open Images is notable for its extensive coverage of object categories and large-scale annotations, making it useful for training and evaluating advanced computer vision models.</p>
<p>These datasets play a crucial role in advancing computer vision research and development by providing standardized benchmarks, training data, and evaluation protocols for various tasks such as object detection, segmentation, and classification. They enable researchers and developers to train and test algorithms on large and diverse datasets, facilitating progress in computer vision technologies.</p>
<p>Since data sets related to jellyfish stings are very scarce, we created synthetic data sets to aid training. Use DCGAN (Deep Convolutional GAN) to synthesize the data set. First, a dataset of real images related to jellyfish stings is collected. The specific steps are as follows: Data preprocessing: Image size: Adjust the image to a uniform size, 64x64 pixels. Normalization: Normalize the image pixel value to the [-1, 1] range, which can be achieved by dividing the pixel value by 255, subtracting 0.5, and then multiplying by 2.DCGAN model architecture: Generator network: Input: Random noise vector, typically with 100 dimensions.Transposed convolution layer: Use ReLU activation function and convolution kernel size of 4x4, gradually increasing the number of channels and image size. Batch normalization: Adding a batch normalization layer after the transposed convolutional layer helps stabilize the training process. Output layer: Use the Tanh activation function to limit the generated image pixel values to the range [-1, 1]. Discriminator network: Input: a real image or a generator-generated image with the same dimensions as the generator output image. Convolutional layer: Use LeakyReLU activation function and appropriate convolution kernel size to gradually reduce the number of channels and image size. Fully connected layer: After flattening the output of the convolutional layer, it is connected to a fully connected layer to output a binary classification result (true or false). Loss function and optimizer: Loss function: Generator loss and discriminator loss use binary cross-entropy loss function. Optimizer: Use the Adam optimizer to optimize model parameters and set the learning rate to 0.0002. Training parameters:Batch Size: The batch size is set to 128. Number of iterations (Epochs): The number of iterations is 10,000. Learning rate decay: The learning rate can be gradually reduced during the training process to help the model stabilize and converge. Generate a synthetic dataset: Once training is complete, the generator network can be used to generate synthetic jellyfish sting target images. To obtain diversity in synthetic data, multiple different random vectors can be used in the generator input. Dataset evaluation: The generated synthetic datasets are evaluated to ensure the resulting image fidelity and similarity to real data. Image quality evaluation indicators such as PSNR and SSIM can be used to evaluate the quality of synthetic data.</p></sec>
<sec>
<title>3.2 Experimental details</title>
<p>In this experiment, We use 8-card nvidia A100-80G for training. our objective is to compare the performance of different models on various metrics and conduct ablation experiments to analyze the factors influencing these metrics. We will focus on the real-time precision detection algorithm for jellyfish stings using adaptive deep learning enhanced by an advanced YOLOv4 framework, as mentioned earlier.</p>
<p>1. Dataset preparation:</p>
<p>Gather a diverse dataset of images or videos containing jellyfish stings, covering various scenarios, lighting conditions, and jellyfish species. Split the dataset into training, validation, and test sets, ensuring that the distribution of data is representative and unbiased.</p>
<p>2. Model selection:</p>
<p>Choose the advanced YOLOv4 framework as the base model for the experiment, considering its real-time performance and accuracy. Optionally, select alternative deep learning architectures, such as Faster R-CNN or SSD, for comparison purposes.</p>
<p>3. Training process:</p>
<p>Initialize the YOLOv4 model with pre-trained weights on a large-scale dataset (e.g., COCO) to leverage transfer learning. Fine-tune the model on the jellyfish stings dataset, adjusting hyperparameters such as learning rate, batch size, and optimization algorithm (e.g., Adam). Monitor and record important metrics during the training process, such as loss, accuracy, and learning curves.</p>
<p>4. Model evaluation:</p>
<p>Evaluate the trained YOLOv4 model on the validation set to assess its performance in terms of precision, recall, and mean average precision (mAP). Measure inference time to evaluate the model&#x00027;s real-time capabilities.</p>
<p>5. Ablation experiments:</p>
<p>Identify specific factors that may influence the algorithm&#x00027;s performance, such as the attention mechanism or PID control. Design ablation experiments by disabling or modifying these factors to analyze their impact on detection precision and real-time performance. Measure and compare the metrics between the original algorithm and the ablated versions, using both quantitative (e.g., mAP, inference time) and qualitative analysis (visual inspection of detection results).</p>
<p>6. Performance analysis:</p>
<p>Compare the performance of different models (e.g., YOLOv4, alternative architectures) on metrics such as precision, recall, mAP, and inference time. Analyze the results of the ablation experiments to understand the influence of specific components or techniques on the algorithm&#x00027;s performance. Present the findings using visualizations, such as performance curves, bar charts, or tables, to facilitate interpretation and comparison.</p>
<p>7. Discussion and conclusion:</p>
<p>Discuss the implications of the experimental results, highlighting the strengths and weaknesses of the proposed algorithm and the impact of different factors on its performance. Draw conclusions based on the analysis and suggest potential avenues for further improvement or research.</p>
<p>Here are the formulas for each metric along with explanations of the variables:</p>
<p>PSNR (Peak signal-to-noise ratio):</p>
<disp-formula id="E6"><label>(6)</label><mml:math id="M9"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">PSNR</mml:mtext><mml:mo>=</mml:mo><mml:mn>10</mml:mn><mml:mo>&#x000B7;</mml:mo><mml:msub><mml:mrow><mml:mo class="qopname">log</mml:mo></mml:mrow><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="true">(</mml:mo><mml:mrow><mml:mfrac><mml:mrow><mml:msup><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mtext class="textrm" mathvariant="normal">MSE</mml:mtext></mml:mrow></mml:mfrac></mml:mrow><mml:mo stretchy="true">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>PSNR measures the quality of a reconstructed or generated image compared to the original image (<xref ref-type="disp-formula" rid="E6">Equation 6</xref>). <italic>L</italic> represents the maximum pixel value (e.g., 255 for 8-bit images). MSE is the mean squared error between the original and reconstructed/generated images.</p>
<p>SSIM (Structural similarity index):</p>
<disp-formula id="E7"><label>(7)</label><mml:math id="M10"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">SSIM</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msubsup><mml:mrow><mml:mi>&#x003C3;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>C</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>SSIM measures the structural similarity between two images (<xref ref-type="disp-formula" rid="E7">Equation 7</xref>). &#x003BC;<sub><italic>x</italic></sub> and &#x003BC;<sub><italic>y</italic></sub> are the means of the original and reconstructed/generated images, respectively. &#x003C3;<sub><italic>x</italic></sub> and &#x003C3;<sub><italic>y</italic></sub> are the standard deviations of the original and reconstructed/generated images, respectively. &#x003C3;<sub><italic>xy</italic></sub> is the covariance between the original and reconstructed/generated images. <italic>C</italic><sub>1</sub> and <italic>C</italic><sub>2</sub> are small constants added for numerical stability.</p>
<p>FID (Frechet inception distance):</p>
<disp-formula id="E8"><label>(8)</label><mml:math id="M11"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">FID</mml:mtext><mml:mo>=</mml:mo><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003BC;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mo>|</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x0002B;</mml:mo><mml:mtext class="textrm" mathvariant="normal">Tr</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003A3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:msub><mml:mrow><mml:mi>&#x003A3;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:mn>2</mml:mn><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>&#x003A3;</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>&#x003A3;</mml:mi></mml:mrow><mml:mrow><mml:mi>y</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:msup></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>FID measures the similarity between the feature distributions of real and generated images (<xref ref-type="disp-formula" rid="E8">Equation 8</xref>). &#x003BC;<sub><italic>x</italic></sub> and &#x003BC;<sub><italic>y</italic></sub> are the means of the feature embeddings of real and generated images, respectively. &#x003A3;<sub><italic>x</italic></sub> and &#x003A3;<sub><italic>y</italic></sub> are the covariance matrices of the feature embeddings of real and generated images, respectively. |&#x000B7;|<sup>2</sup> represents the squared Euclidean distance, and Tr(&#x000B7;) is the trace operator.</p>
<p>IS (Inception Score):</p>
<disp-formula id="E9"><label>(9)</label><mml:math id="M12"><mml:mtable class="eqnarray" columnalign="left"><mml:mtr><mml:mtd><mml:mtext class="textrm" mathvariant="normal">IS</mml:mtext><mml:mo>=</mml:mo><mml:mo class="qopname">exp</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x1D53C;</mml:mi><mml:mstyle mathvariant="bold"><mml:mtext>x</mml:mtext></mml:mstyle><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>D</mml:mi><mml:mtext class="textrm" mathvariant="normal">KL</mml:mtext><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle><mml:mo>||</mml:mo><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mstyle mathvariant="bold"><mml:mtext>y</mml:mtext></mml:mstyle></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>IS measures the quality and diversity of generated images (<xref ref-type="disp-formula" rid="E9">Equation 9</xref>). <bold>x</bold> represents the generated images. <bold>y</bold> is the class probability distribution predicted by an Inception model. <italic>p</italic>(<bold>y</bold>) is the marginal class distribution of the generated images. <italic>D</italic><sub>KL</sub>(&#x000B7;) denotes the Kullback-Leibler divergence.</p>
<p><xref ref-type="table" rid="T9">Algorithm 1</xref> represents the training process of the proposed model (<xref ref-type="table" rid="T9">Equations 10&#x02013;15</xref>).</p>
<table-wrap position="float" id="T9">
<label>Algorithm 1</label>
<caption><p>Training process for YAM-PID Net.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-i0001.tif"/>
</table-wrap>
</sec>
<sec>
<title>3.3 Experimental results and analysis</title>
<p><xref ref-type="table" rid="T1">Tables 1</xref>, <xref ref-type="table" rid="T2">2</xref> and <xref ref-type="fig" rid="F5">Figure 5</xref> presents the performance comparison of our designed adaptive intelligent robot detection algorithm on different datasets. Our method utilizes an improved version of the YOLOv4 object detection framework, combined with attention mechanisms and PID control algorithm, to achieve real-time and accurate detection of sea Jellyfish injuries in complex environments. The following are the main findings and conclusions of the experimental results. Our method exhibits a relatively low number of model parameters and floating-point operations, ensuring a lightweight model suitable for embedded devices and real-time applications. In terms of inference time and training time, our method outperforms other approaches, making it more practical for real-time applications. Furthermore, our method demonstrates competitive performance on various datasets, showcasing its adaptability to different scenarios and tasks. Our method shows significant advantages in lightweight design and real-time performance, along with excellent adaptability across multiple datasets. By incorporating the improved YOLOv4, attention mechanisms, and PID control algorithm, our proposed adaptive intelligent robot detection algorithm excels in real-time and precise detection of sea Jellyfish injuries, making it one of the most suitable solutions for the current task. Our algorithm not only demonstrates competitive performance but also holds advantages in lightweight design and real-time efficiency, providing reliable support for intelligent robots in complex environments.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Comparison of different models on COCO and Pascal VOC datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>References</bold></th>
<th valign="top" align="left" colspan="4"><bold>COCO dataset (Sharma</bold>, <xref ref-type="bibr" rid="B20"><bold>2021</bold></xref><bold>)</bold></th>
<th valign="top" align="left" colspan="4"><bold>Pascal VOC dataset (Tong and Wu</bold>, <xref ref-type="bibr" rid="B22"><bold>2023</bold></xref><bold>)</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="left"><bold>Parameters (M)</bold></td>
<td valign="top" align="left"><bold>Flops (G)</bold></td>
<td valign="top" align="left"><bold>Inference time (ms)</bold></td>
<td valign="top" align="left"><bold>Training time (s)</bold></td>
<td valign="top" align="left"><bold>Parameters (M)</bold></td>
<td valign="top" align="left"><bold>Flops (G)</bold></td>
<td valign="top" align="left"><bold>Inference time (ms)</bold></td>
<td valign="top" align="left"><bold>Training time (s)</bold></td>
</tr> <tr>
<td valign="top" align="left">(Gao M. et al., <xref ref-type="bibr" rid="B6">2021</xref>)</td>
<td valign="top" align="left">245.39</td>
<td valign="top" align="left">356.78</td>
<td valign="top" align="left">343.79</td>
<td valign="top" align="left">319.34</td>
<td valign="top" align="left">389.83</td>
<td valign="top" align="left">323.01</td>
<td valign="top" align="left">255.72</td>
<td valign="top" align="left">242.07</td>
</tr> <tr>
<td valign="top" align="left">(Zhao et al., <xref ref-type="bibr" rid="B32">2020</xref>)</td>
<td valign="top" align="left">399.92</td>
<td valign="top" align="left">256.82</td>
<td valign="top" align="left">203.67</td>
<td valign="top" align="left">332.95</td>
<td valign="top" align="left">340.40</td>
<td valign="top" align="left">375.07</td>
<td valign="top" align="left">306.32</td>
<td valign="top" align="left">392.20</td>
</tr> <tr>
<td valign="top" align="left">(Yu et al., <xref ref-type="bibr" rid="B28">2024</xref>)</td>
<td valign="top" align="left">296.53</td>
<td valign="top" align="left">320.64</td>
<td valign="top" align="left">375.42</td>
<td valign="top" align="left">226.84</td>
<td valign="top" align="left">224.57</td>
<td valign="top" align="left">358.37</td>
<td valign="top" align="left">384.04</td>
<td valign="top" align="left">387.44</td>
</tr> <tr>
<td valign="top" align="left">(Yun et al., <xref ref-type="bibr" rid="B29">2022</xref>)</td>
<td valign="top" align="left">347.19</td>
<td valign="top" align="left">314.59</td>
<td valign="top" align="left">332.30</td>
<td valign="top" align="left">268.44</td>
<td valign="top" align="left">215.26</td>
<td valign="top" align="left">237.65</td>
<td valign="top" align="left">233.86</td>
<td valign="top" align="left">201.18</td>
</tr> <tr>
<td valign="top" align="left">(Tan et al., <xref ref-type="bibr" rid="B21">2021</xref>)</td>
<td valign="top" align="left">252.65</td>
<td valign="top" align="left">272.84</td>
<td valign="top" align="left">261.66</td>
<td valign="top" align="left">346.64</td>
<td valign="top" align="left">348.86</td>
<td valign="top" align="left">341.26</td>
<td valign="top" align="left">392.52</td>
<td valign="top" align="left">257.45</td>
</tr> <tr>
<td valign="top" align="left">(Dai et al., <xref ref-type="bibr" rid="B4">2021</xref>)</td>
<td valign="top" align="left">370.44</td>
<td valign="top" align="left">239.58</td>
<td valign="top" align="left">297.10</td>
<td valign="top" align="left">226.78</td>
<td valign="top" align="left">342.71</td>
<td valign="top" align="left">221.42</td>
<td valign="top" align="left">251.90</td>
<td valign="top" align="left">326.18</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="left">123.32</td>
<td valign="top" align="left">118.40</td>
<td valign="top" align="left">132.70</td>
<td valign="top" align="left">110.85</td>
<td valign="top" align="left">144.23</td>
<td valign="top" align="left">137.33</td>
<td valign="top" align="left">143.61</td>
<td valign="top" align="left">105.30</td>
</tr></tbody>
</table>
</table-wrap>

<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Comparison of different models on KITTI and Open Images datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="4"><bold>KITTI dataset (Al-refai and Al-refai</bold>, <xref ref-type="bibr" rid="B1"><bold>2020</bold></xref><bold>)</bold></th>
<th valign="top" align="center" colspan="4"><bold>Open images dataset (Veit et al.</bold>, <xref ref-type="bibr" rid="B23"><bold>2017</bold></xref><bold>)</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="center"><bold>Parameters (M)</bold></td>
<td valign="top" align="center"><bold>Flops (G)</bold></td>
<td valign="top" align="left"><bold>Inference time (ms)</bold></td>
<td valign="top" align="left"><bold>Training time (s)</bold></td>
<td valign="top" align="left"><bold>Parameters (M)</bold></td>
<td valign="top" align="left"><bold>Flops (G)</bold></td>
<td valign="top" align="left"><bold>Inference time (ms)</bold></td>
<td valign="top" align="left"><bold>Training time (s)</bold></td>
</tr> <tr>
<td valign="top" align="left">Gao et al.</td>
<td valign="top" align="center">242.70</td>
<td valign="top" align="center">320.81</td>
<td valign="top" align="left">275.98</td>
<td valign="top" align="left">221.11</td>
<td valign="top" align="left">252.39</td>
<td valign="top" align="left">336.39</td>
<td valign="top" align="left">240.74</td>
<td valign="top" align="left">497.90</td>
</tr> <tr>
<td valign="top" align="left">Zhao et al.</td>
<td valign="top" align="center">309.40</td>
<td valign="top" align="center">260.76</td>
<td valign="top" align="left">328.22</td>
<td valign="top" align="left">352.15</td>
<td valign="top" align="left">265.77</td>
<td valign="top" align="left">341.26</td>
<td valign="top" align="left">340.54</td>
<td valign="top" align="left">592.05</td>
</tr> <tr>
<td valign="top" align="left">Yu et al.</td>
<td valign="top" align="center">347.34</td>
<td valign="top" align="center">374.27</td>
<td valign="top" align="left">395.33</td>
<td valign="top" align="left">286.44</td>
<td valign="top" align="left">267.94</td>
<td valign="top" align="left">354.57</td>
<td valign="top" align="left">250.22</td>
<td valign="top" align="left">456.36</td>
</tr> <tr>
<td valign="top" align="left">Yun et al.</td>
<td valign="top" align="center">289.79</td>
<td valign="top" align="center">273.63</td>
<td valign="top" align="left">214.21</td>
<td valign="top" align="left">314.67</td>
<td valign="top" align="left">233.95</td>
<td valign="top" align="left">355.40</td>
<td valign="top" align="left">238.47</td>
<td valign="top" align="left">358.33</td>
</tr> <tr>
<td valign="top" align="left">Tan et al.</td>
<td valign="top" align="center">396.47</td>
<td valign="top" align="center">363.19</td>
<td valign="top" align="left">304.74</td>
<td valign="top" align="left">389.45</td>
<td valign="top" align="left">297.36</td>
<td valign="top" align="left">215.87</td>
<td valign="top" align="left">398.27</td>
<td valign="top" align="left">394.14</td>
</tr> <tr>
<td valign="top" align="left">Dai et al.</td>
<td valign="top" align="center">274.82</td>
<td valign="top" align="center">323.95</td>
<td valign="top" align="left">276.18</td>
<td valign="top" align="left">327.39</td>
<td valign="top" align="left">340.94</td>
<td valign="top" align="left">257.29</td>
<td valign="top" align="left">311.82</td>
<td valign="top" align="left">276.92</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">210.50</td>
<td valign="top" align="center">176.41</td>
<td valign="top" align="left">188.73</td>
<td valign="top" align="left">114.90</td>
<td valign="top" align="left">146.18</td>
<td valign="top" align="left">214.99</td>
<td valign="top" align="left">152.76</td>
<td valign="top" align="left">232.85</td>
</tr></tbody>
</table>
</table-wrap>

<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Comparison of different models on different datasets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0005.tif"/>
</fig>




<p><xref ref-type="table" rid="T3">Tables 3</xref>, <xref ref-type="table" rid="T4">4</xref> and <xref ref-type="fig" rid="F6">Figure 6</xref> show the performance comparison of our designed model on different data sets. Experimental results show that our model DCGAN (Deep Convolutional GAN) has advantages in image quality, showing higher PSNR, SSIM and IS values. Furthermore, our model also achieves the best performance in terms of image diversity and realism, as shown by the lowest FID score. Compared with other compared methods, our model shows excellent performance on all metrics, demonstrating a higher level of overall performance. Therefore, it can provide an accurate data set for jellyfish sting training.</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Comparison of different models on COCO and Pascal VOC datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left" colspan="4"><bold>COCO dataset</bold></th>
<th valign="top" align="left" colspan="4"><bold>Pascal VOC dataset</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="left"><bold>PSNR</bold>&#x02191;</td>
<td valign="top" align="left"><bold>FID</bold>&#x02193;</td>
<td valign="top" align="left"><bold>SSIM</bold>&#x02191;</td>
<td valign="top" align="left"><bold>IS</bold>&#x02191;</td>
<td valign="top" align="left"><bold>PSNR</bold>&#x02191;</td>
<td valign="top" align="left"><bold>FID</bold>&#x02193;</td>
<td valign="top" align="left"><bold>SSIM</bold>&#x02191;</td>
<td valign="top" align="left"><bold>IS</bold>&#x02191;</td>
</tr> <tr>
<td valign="top" align="left">Gao et al.</td>
<td valign="top" align="left">25.71</td>
<td valign="top" align="left">18.32</td>
<td valign="top" align="left">0.62</td>
<td valign="top" align="left">9.79</td>
<td valign="top" align="left">27.12</td>
<td valign="top" align="left">18.45</td>
<td valign="top" align="left">0.72</td>
<td valign="top" align="left">11.35</td>
</tr> <tr>
<td valign="top" align="left">Zhao et al.</td>
<td valign="top" align="left">24.92</td>
<td valign="top" align="left">27.26</td>
<td valign="top" align="left">0.67</td>
<td valign="top" align="left">9.53</td>
<td valign="top" align="left">22.89</td>
<td valign="top" align="left">25.65</td>
<td valign="top" align="left">0.62</td>
<td valign="top" align="left">11.56</td>
</tr> <tr>
<td valign="top" align="left">Yu et al.</td>
<td valign="top" align="left">26.65</td>
<td valign="top" align="left">23.41</td>
<td valign="top" align="left">0.61</td>
<td valign="top" align="left">10.64</td>
<td valign="top" align="left">26.86</td>
<td valign="top" align="left">11.12</td>
<td valign="top" align="left">0.75</td>
<td valign="top" align="left">11.85</td>
</tr> <tr>
<td valign="top" align="left">Yun et al.</td>
<td valign="top" align="left">27.38</td>
<td valign="top" align="left">20.18</td>
<td valign="top" align="left">0.61</td>
<td valign="top" align="left">8.95</td>
<td valign="top" align="left">22.14</td>
<td valign="top" align="left">21.04</td>
<td valign="top" align="left">0.59</td>
<td valign="top" align="left">9.69</td>
</tr> <tr>
<td valign="top" align="left">Tan et al.</td>
<td valign="top" align="left">26.28</td>
<td valign="top" align="left">18.99</td>
<td valign="top" align="left">0.64</td>
<td valign="top" align="left">11.8</td>
<td valign="top" align="left">26.37</td>
<td valign="top" align="left">9.37</td>
<td valign="top" align="left">0.73</td>
<td valign="top" align="left">11.49</td>
</tr> <tr>
<td valign="top" align="left">Dai et al.</td>
<td valign="top" align="left">27.31</td>
<td valign="top" align="left">26.7</td>
<td valign="top" align="left">0.7</td>
<td valign="top" align="left">10.4</td>
<td valign="top" align="left">23.01</td>
<td valign="top" align="left">10.06</td>
<td valign="top" align="left">0.59</td>
<td valign="top" align="left">8.79</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="left">32.18</td>
<td valign="top" align="left">6.6</td>
<td valign="top" align="left">0.84</td>
<td valign="top" align="left">11.95</td>
<td valign="top" align="left">29.81</td>
<td valign="top" align="left">6.08</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">12.34</td>
</tr></tbody>
</table>
</table-wrap>

<table-wrap position="float" id="T4">
<label>Table 4</label>
<caption><p>Comparison of different models on KITTI and Open Images datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left" colspan="4"><bold>KITTI dataset</bold></th>
<th valign="top" align="left" colspan="4"><bold>Open images dataset</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="left"><bold>PSNR</bold>&#x02191;</td>
<td valign="top" align="left"><bold>FID</bold>&#x02193;</td>
<td valign="top" align="left"><bold>SSIM</bold>&#x02191;</td>
<td valign="top" align="left"><bold>IS</bold>&#x02191;</td>
<td valign="top" align="left"><bold>PSNR</bold>&#x02191;</td>
<td valign="top" align="left"><bold>FID</bold>&#x02193;</td>
<td valign="top" align="left"><bold>SSIM</bold>&#x02191;</td>
<td valign="top" align="left"><bold>IS</bold>&#x02191;</td>
</tr> <tr>
<td valign="top" align="left">Gao et al.</td>
<td valign="top" align="left">24.25</td>
<td valign="top" align="left">9.79</td>
<td valign="top" align="left">0.64</td>
<td valign="top" align="left">11.41</td>
<td valign="top" align="left">25.49</td>
<td valign="top" align="left">19.34</td>
<td valign="top" align="left">0.64</td>
<td valign="top" align="left">9.57</td>
</tr> <tr>
<td valign="top" align="left">Zhao et al.</td>
<td valign="top" align="left">23.87</td>
<td valign="top" align="left">21.66</td>
<td valign="top" align="left">0.57</td>
<td valign="top" align="left">11.69</td>
<td valign="top" align="left">25.75</td>
<td valign="top" align="left">19.84</td>
<td valign="top" align="left">0.53</td>
<td valign="top" align="left">10.94</td>
</tr> <tr>
<td valign="top" align="left">Yu et al.</td>
<td valign="top" align="left">24.82</td>
<td valign="top" align="left">18.33</td>
<td valign="top" align="left">0.75</td>
<td valign="top" align="left">10.14</td>
<td valign="top" align="left">23.89</td>
<td valign="top" align="left">24.23</td>
<td valign="top" align="left">0.65</td>
<td valign="top" align="left">8.25</td>
</tr> <tr>
<td valign="top" align="left">Yun et al.</td>
<td valign="top" align="left">27.15</td>
<td valign="top" align="left">25.20</td>
<td valign="top" align="left">0.53</td>
<td valign="top" align="left">8.73</td>
<td valign="top" align="left">24.38</td>
<td valign="top" align="left">12.57</td>
<td valign="top" align="left">0.57</td>
<td valign="top" align="left">9.55</td>
</tr> <tr>
<td valign="top" align="left">Tan et al.</td>
<td valign="top" align="left">23.47</td>
<td valign="top" align="left">10.36</td>
<td valign="top" align="left">0.55</td>
<td valign="top" align="left">8.23</td>
<td valign="top" align="left">21.61</td>
<td valign="top" align="left">10.91</td>
<td valign="top" align="left">0.57</td>
<td valign="top" align="left">9.26</td>
</tr> <tr>
<td valign="top" align="left">Dai et al.</td>
<td valign="top" align="left">26.99</td>
<td valign="top" align="left">12.02</td>
<td valign="top" align="left">0.66</td>
<td valign="top" align="left">8.52</td>
<td valign="top" align="left">22.62</td>
<td valign="top" align="left">13.24</td>
<td valign="top" align="left">0.65</td>
<td valign="top" align="left">10.98</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="left">31.18</td>
<td valign="top" align="left">8.04</td>
<td valign="top" align="left">0.77</td>
<td valign="top" align="left">12.06</td>
<td valign="top" align="left">29.80</td>
<td valign="top" align="left">7.08</td>
<td valign="top" align="left">0.83</td>
<td valign="top" align="left">12.23</td>
</tr></tbody>
</table>
</table-wrap>



<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Comparison of different models on different data sets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0006.tif"/>
</fig>




<p><xref ref-type="table" rid="T5">Tables 5</xref>, <xref ref-type="table" rid="T6">6</xref> and <xref ref-type="fig" rid="F7">Figure 7</xref> presents the results of our conducted experiments on the Advanced YOLOv4 module, comparing the performance of different methods on various datasets. Our model incorporates improved attention mechanisms and PID control algorithms, combined with the Advanced YOLOv4 module, aiming to achieve lightweight design and high performance. The experimental results demonstrate that our model exhibits lower model parameter count and floating-point operation count, successfully achieving the goal of lightweight design. Additionally, our model achieves favorable results in terms of inference time and training time, demonstrating high real-time performance and training efficiency. Compared to traditional methods like R-CNN and other lightweight models such as EfficientDet, our model outperforms in all evaluated metrics, showcasing superior performance and efficiency. By introducing improved attention mechanisms and PID control algorithms, our model demonstrates excellent performance in complex detection tasks, providing reliable support for practical applications. Overall, our Advanced YOLOv4 module stands as one of the most competitive and practical solutions, with vast potential for real-world applications.</p>
<table-wrap position="float" id="T5">
<label>Table 5</label>
<caption><p>Ablation experiments on advanced YOLOv4 module for COCO and Pascal VOC datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="4"><bold>COCO dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>Pascal VOC dataset</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Parameters (M)</bold></th>
<th valign="top" align="center"><bold>Flops (G)</bold></th>
<th valign="top" align="left"><bold>Inference time (ms)</bold></th>
<th valign="top" align="left"><bold>Training time (s)</bold></th>
<th valign="top" align="center"><bold>Parameters (M)</bold></th>
<th valign="top" align="center"><bold>Flops (G)</bold></th>
<th valign="top" align="left"><bold>Inference time (ms)</bold></th>
<th valign="top" align="left"><bold>Training time (s)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">yolov3</td>
<td valign="top" align="center">392.34</td>
<td valign="top" align="center">377.61</td>
<td valign="top" align="left">277.29</td>
<td valign="top" align="left">211.63</td>
<td valign="top" align="center">205.60</td>
<td valign="top" align="center">322.27</td>
<td valign="top" align="left">268.19</td>
<td valign="top" align="left">221.68</td>
</tr> <tr>
<td valign="top" align="left">R-CNN</td>
<td valign="top" align="center">319.87</td>
<td valign="top" align="center">277.98</td>
<td valign="top" align="left">370.84</td>
<td valign="top" align="left">243.65</td>
<td valign="top" align="center">290.34</td>
<td valign="top" align="center">299.16</td>
<td valign="top" align="left">254.89</td>
<td valign="top" align="left">301.40</td>
</tr> <tr>
<td valign="top" align="left">EfficientDet</td>
<td valign="top" align="center">334.58</td>
<td valign="top" align="center">344.05</td>
<td valign="top" align="left">312.01</td>
<td valign="top" align="left">222.72</td>
<td valign="top" align="center">316.15</td>
<td valign="top" align="center">205.06</td>
<td valign="top" align="left">276.13</td>
<td valign="top" align="left">334.03</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">180.61</td>
<td valign="top" align="center">144.39</td>
<td valign="top" align="left">220.57</td>
<td valign="top" align="left">138.77</td>
<td valign="top" align="center">171.89</td>
<td valign="top" align="center">163.93</td>
<td valign="top" align="left">203.47</td>
<td valign="top" align="left">192.27</td>
</tr></tbody>
</table>
</table-wrap>


<table-wrap position="float" id="T6">
<label>Table 6</label>
<caption><p>Ablation experiments on advanced YOLOv4 module for KITTI and Open Images datasets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Method</bold></th>
<th valign="top" align="center" colspan="4"><bold>KITTI dataset</bold></th>
<th valign="top" align="center" colspan="4"><bold>Open images dataset</bold></th>
</tr>
<tr style="background-color:#919498;color:#ffffff">
<th/>
<th valign="top" align="center"><bold>Parameters (M)</bold></th>
<th valign="top" align="center"><bold>Flops (G)</bold></th>
<th valign="top" align="left"><bold>Inference time (ms)</bold></th>
<th valign="top" align="left"><bold>Training time (s)</bold></th>
<th valign="top" align="center"><bold>Parameters (M)</bold></th>
<th valign="top" align="center"><bold>Flops (G)</bold></th>
<th valign="top" align="left"><bold>Inference time (ms)</bold></th>
<th valign="top" align="left"><bold>Training time (s)</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">yolov3</td>
<td valign="top" align="center">294.05</td>
<td valign="top" align="center">315.03</td>
<td valign="top" align="left">296.83</td>
<td valign="top" align="left">343.84</td>
<td valign="top" align="center">397.93</td>
<td valign="top" align="center">244.49</td>
<td valign="top" align="left">319.61</td>
<td valign="top" align="left">246.65</td>
</tr> <tr>
<td valign="top" align="left">R-CNN</td>
<td valign="top" align="center">290.31</td>
<td valign="top" align="center">213.21</td>
<td valign="top" align="left">250.09</td>
<td valign="top" align="left">221.42</td>
<td valign="top" align="center">380.05</td>
<td valign="top" align="center">234.04</td>
<td valign="top" align="left">377.30</td>
<td valign="top" align="left">240.38</td>
</tr> <tr>
<td valign="top" align="left">EfficientDet</td>
<td valign="top" align="center">267.81</td>
<td valign="top" align="center">285.12</td>
<td valign="top" align="left">271.03</td>
<td valign="top" align="left">312.30</td>
<td valign="top" align="center">361.09</td>
<td valign="top" align="center">334.80</td>
<td valign="top" align="left">298.40</td>
<td valign="top" align="left">224.82</td>
</tr>
<tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="center">130.53</td>
<td valign="top" align="center">140.20</td>
<td valign="top" align="left">136.60</td>
<td valign="top" align="left">208.54</td>
<td valign="top" align="center">197.80</td>
<td valign="top" align="center">179.72</td>
<td valign="top" align="left">119.45</td>
<td valign="top" align="left">186.46</td>
</tr></tbody>
</table>
</table-wrap>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>Ablation experiments on Advanced YOLOv4 module.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0007.tif"/>
</fig>



<p><xref ref-type="table" rid="T7">Table 7</xref> and <xref ref-type="fig" rid="F8">Figure 8</xref> show the experimental results we conducted on the attention mechanism module, comparing the performance of different methods on various data sets. Comparison methods and principles: No-AM (No Attention Mechanism): Baseline model without any attention mechanism. Self-AM (Self-Attention Mechanism): Use self-attention mechanism to capture long-range dependencies in images. Cross-AM (Cross-Attention Mechanism): Use the cross-attention mechanism to handle the correlation between different areas. Our: Our proposed model incorporates an improved attention mechanism and aims to improve the performance of image reconstruction and generation tasks. By introducing the attention mechanism module, DCGAN (Deep Convolutional GAN) has achieved significant performance improvements in image generation and reconstruction tasks. Our model shows outstanding advantages in image quality, distribution similarity, structural similarity, and diversity and quality. This makes it a good candidate for generating an ideal experimental data set for jellyfish stings.</p>
<table-wrap position="float" id="T7">
<label>Table 7</label>
<caption><p>Ablation experiments on attention mechanism module.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left" colspan="16"><bold>Datasets</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="left" colspan="4"><bold>COCO dataset</bold></td>
<td valign="top" align="left" colspan="4"><bold>Pascal VOC dataset</bold></td>
<td valign="top" align="left" colspan="4"><bold>KITTI dataset</bold></td>
<td valign="top" align="left" colspan="4"><bold>Open images dataset</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left"><bold>PSNR</bold>&#x02191;</td>
<td valign="top" align="left"><bold>FID</bold>&#x02193;</td>
<td valign="top" align="left"><bold>SSIM</bold>&#x02191;</td>
<td valign="top" align="left"><bold>IS</bold>&#x02191;</td>
<td valign="top" align="left"><bold>PSNR</bold>&#x02191;</td>
<td valign="top" align="left"><bold>FID</bold>&#x02193;</td>
<td valign="top" align="left"><bold>SSIM</bold>&#x02191;</td>
<td valign="top" align="left"><bold>IS</bold>&#x02191;</td>
<td valign="top" align="left"><bold>PSNR</bold>&#x02191;</td>
<td valign="top" align="left"><bold>FID</bold>&#x02193;</td>
<td valign="top" align="left"><bold>SSIM</bold>&#x02191;</td>
<td valign="top" align="left"><bold>IS</bold>&#x02191;</td>
<td valign="top" align="left"><bold>PSNR</bold>&#x02191;</td>
<td valign="top" align="left"><bold>FID</bold>&#x02193;</td>
<td valign="top" align="left"><bold>SSIM</bold>&#x02191;</td>
<td valign="top" align="left"><bold>IS</bold>&#x02191;</td>
</tr> <tr>
<td valign="top" align="left">No-AM</td>
<td valign="top" align="left">24.71</td>
<td valign="top" align="left">15.35</td>
<td valign="top" align="left">0.65</td>
<td valign="top" align="left">10.65</td>
<td valign="top" align="left">22.79</td>
<td valign="top" align="left">20.37</td>
<td valign="top" align="left">0.54</td>
<td valign="top" align="left">9.86</td>
<td valign="top" align="left">25.15</td>
<td valign="top" align="left">13.38</td>
<td valign="top" align="left">0.72</td>
<td valign="top" align="left">11.9</td>
<td valign="top" align="left">22.17</td>
<td valign="top" align="left">8.93</td>
<td valign="top" align="left">0.57</td>
<td valign="top" align="left">8.88</td>
</tr> <tr>
<td valign="top" align="left">Self-AM</td>
<td valign="top" align="left">23.8</td>
<td valign="top" align="left">15.24</td>
<td valign="top" align="left">0.64</td>
<td valign="top" align="left">9.88</td>
<td valign="top" align="left">21.9</td>
<td valign="top" align="left">12.78</td>
<td valign="top" align="left">0.58</td>
<td valign="top" align="left">10.59</td>
<td valign="top" align="left">26.87</td>
<td valign="top" align="left">14.62</td>
<td valign="top" align="left">0.62</td>
<td valign="top" align="left">11.05</td>
<td valign="top" align="left">22.39</td>
<td valign="top" align="left">18.17</td>
<td valign="top" align="left">0.75</td>
<td valign="top" align="left">11.68</td>
</tr> <tr>
<td valign="top" align="left">Cross-AM</td>
<td valign="top" align="left">27.22</td>
<td valign="top" align="left">26.84</td>
<td valign="top" align="left">0.63</td>
<td valign="top" align="left">10.58</td>
<td valign="top" align="left">21.44</td>
<td valign="top" align="left">13.2</td>
<td valign="top" align="left">0.59</td>
<td valign="top" align="left">11.45</td>
<td valign="top" align="left">25.26</td>
<td valign="top" align="left">14.11</td>
<td valign="top" align="left">0.7</td>
<td valign="top" align="left">9.64</td>
<td valign="top" align="left">22.47</td>
<td valign="top" align="left">16.2</td>
<td valign="top" align="left">0.54</td>
<td valign="top" align="left">9.42</td>
</tr> <tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="left">31.83</td>
<td valign="top" align="left">7.35</td>
<td valign="top" align="left">0.79</td>
<td valign="top" align="left">11.99</td>
<td valign="top" align="left">32.12</td>
<td valign="top" align="left">8.29</td>
<td valign="top" align="left">0.82</td>
<td valign="top" align="left">12.13</td>
<td valign="top" align="left">28.59</td>
<td valign="top" align="left">8.2</td>
<td valign="top" align="left">0.84</td>
<td valign="top" align="left">12.19</td>
<td valign="top" align="left">30.81</td>
<td valign="top" align="left">6.63</td>
<td valign="top" align="left">0.79</td>
<td valign="top" align="left">12.12</td>
</tr></tbody>
</table>
</table-wrap>

<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Ablation experiments on attention mechanism module.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0008.tif"/>
</fig>



<p>In <xref ref-type="table" rid="T8">Table 8</xref> and <xref ref-type="fig" rid="F9">Figure 9</xref> we present a comparison of the results of a series of experiments performed on two synthetic datasets. Datasets: We used two synthetic datasets named Synthetic Dataset 1 and Synthetic Dataset 2. These datasets are generated to simulate specific tasks. Indicator description: Accuracy: The proportion of samples correctly classified by the classification model. Recall: The proportion of true positive samples that are correctly predicted as positive. F1 Score: An indicator that considers both precision and recall and is used to evaluate the performance of a classification model. AUC: The area under the receiver operating characteristic (ROC) curve, used to evaluate the performance of a binary classification model. Comparing methods: We compared our method with the methods proposed by Gao et al., Zhao et al., Yu et al., Yun et al., Tan et al., and Dai et al. These methods were previously proposed for similar tasks and are used to validate the performance of our method on synthetic datasets. Our method: In the table, our method is labeled &#x0201C;our&#x0201D;. As can be seen from the results, our method achieves the best performance on both synthetic datasets. Result analysis: Our method achieved an accuracy of 97.9% and 98.44% on two data sets, significantly better than other methods (<xref ref-type="fig" rid="F10">Figures 10</xref>, <xref ref-type="fig" rid="F11">11</xref>). Furthermore, our method shows excellent performance in terms of recall, F1 score, and AUC, indicating the effectiveness and robustness of our model on synthetic datasets. Can fully carry out jellyfish sting detection work.</p>
<table-wrap position="float" id="T8">
<label>Table 8</label>
<caption><p>Comparative results on synthetic data sets.</p></caption>
<table frame="box" rules="all">
<thead>
<tr style="background-color:#919498;color:#ffffff">
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="left" colspan="8"><bold>Datasets</bold></th>
</tr>
</thead>
<tbody>
<tr style="background-color:#919498;color:#ffffff">
<td/>
<td valign="top" align="left" colspan="4"><bold>Synthetic dataset 1</bold></td>
<td valign="top" align="left" colspan="4"><bold>Synthetic dataset 2</bold></td>
</tr>
 <tr>
<td/>
<td valign="top" align="left"><bold>Accuracy</bold></td>
<td valign="top" align="left"><bold>Recall</bold></td>
<td valign="top" align="left"><bold>F1 sorce</bold></td>
<td valign="top" align="left"><bold>AUC</bold></td>
<td valign="top" align="left"><bold>Accuracy</bold></td>
<td valign="top" align="left"><bold>Recall</bold></td>
<td valign="top" align="left"><bold>F1 sorce</bold></td>
<td valign="top" align="left"><bold>AUC</bold></td>
</tr> <tr>
<td valign="top" align="left">Gao et al.</td>
<td valign="top" align="left">87</td>
<td valign="top" align="left">88.48</td>
<td valign="top" align="left">88.26</td>
<td valign="top" align="left">83.85</td>
<td valign="top" align="left">86.31</td>
<td valign="top" align="left">85.73</td>
<td valign="top" align="left">89.62</td>
<td valign="top" align="left">86.82</td>
</tr> <tr>
<td valign="top" align="left">Zhao et al.</td>
<td valign="top" align="left">86.28</td>
<td valign="top" align="left">88.31</td>
<td valign="top" align="left">84.6</td>
<td valign="top" align="left">89.48</td>
<td valign="top" align="left">96.05</td>
<td valign="top" align="left">91.47</td>
<td valign="top" align="left">90.27</td>
<td valign="top" align="left">91.86</td>
</tr> <tr>
<td valign="top" align="left">Yu et al.</td>
<td valign="top" align="left">92.78</td>
<td valign="top" align="left">87.24</td>
<td valign="top" align="left">85.87</td>
<td valign="top" align="left">83.95</td>
<td valign="top" align="left">90.23</td>
<td valign="top" align="left">87.67</td>
<td valign="top" align="left">84.62</td>
<td valign="top" align="left">89.11</td>
</tr> <tr>
<td valign="top" align="left">Yun et al.</td>
<td valign="top" align="left">90.7</td>
<td valign="top" align="left">89.36</td>
<td valign="top" align="left">87.7</td>
<td valign="top" align="left">88.93</td>
<td valign="top" align="left">90.36</td>
<td valign="top" align="left">85.6</td>
<td valign="top" align="left">88.36</td>
<td valign="top" align="left">90.35</td>
</tr> <tr>
<td valign="top" align="left">Tan et al.</td>
<td valign="top" align="left">95.31</td>
<td valign="top" align="left">93.3</td>
<td valign="top" align="left">86.91</td>
<td valign="top" align="left">85.03</td>
<td valign="top" align="left">92.55</td>
<td valign="top" align="left">87.07</td>
<td valign="top" align="left">88.81</td>
<td valign="top" align="left">88.2</td>
</tr> <tr>
<td valign="top" align="left">Dai et al.</td>
<td valign="top" align="left">95.52</td>
<td valign="top" align="left">91.41</td>
<td valign="top" align="left">88.28</td>
<td valign="top" align="left">91.86</td>
<td valign="top" align="left">96.13</td>
<td valign="top" align="left">92.02</td>
<td valign="top" align="left">87.19</td>
<td valign="top" align="left">86.65</td>
</tr> <tr>
<td valign="top" align="left">Ours</td>
<td valign="top" align="left">97.9</td>
<td valign="top" align="left">94.54</td>
<td valign="top" align="left">92.44</td>
<td valign="top" align="left">95.51</td>
<td valign="top" align="left">98.44</td>
<td valign="top" align="left">94.35</td>
<td valign="top" align="left">92.89</td>
<td valign="top" align="left">96.59</td>
</tr></tbody>
</table>
</table-wrap>

<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>Comparative results on synthetic data sets.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0009.tif"/>
</fig>

<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>For the jellyfish broken target detection results of the proposed method, each score represents the recognition accuracy of different broken targets, and the accuracy is as high as 0.9956&#x02013;0.9634. This shows the high efficiency and accuracy of the detection system in identifying different levels of fractures.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0010.tif"/>
</fig>

<fig id="F11" position="float">
<label>Figure 11</label>
<caption><p>Pictured are four different t-SNE visualizations of the proposed method, each using a different &#x0201C;perplexity&#x0201D; value of 5, 30, 50, and 100. The figure shows that the proposed method can distinguish different categories of data well.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fnbot-18-1375886-g0011.tif"/>
</fig>




</sec></sec>
<sec id="s4">
<title>4 Conclusion and discussion</title>
<p>In this study, we aimed to address key issues in image generation and reconstruction tasks by improving the quality, diversity, and structural similarity of generated images. We focused on various datasets, including COCO Dataset, Pascal VOC Dataset, KITTI Dataset, and Open Images Dataset, to comprehensively evaluate the performance of our method in different scenarios. We proposed two key improvement modules: an attention mechanism introduced in the Advanced YOLOv4 module and an improved attention mechanism introduced in the general image generation model. These modules aimed to better handle long-range dependencies and region correlations, thereby enhancing the performance of image generation tasks. Our comparative analysis reveals that our approach significantly outperforms classical methods across multiple datasets. Specifically, on the COCO Dataset, our method achieved a PSNR of 32.18, a low FID of 6.6, an SSIM of 0.84, and an IS of 11.95, indicating superior image quality and consistency. Similarly, on the Pascal VOC Dataset, we noted improvements with a PSNR of 29.81, FID of 6.08, SSIM of 0.83, and IS of 12.34. This trend of enhanced performance continues across the KITTI and Open Images Datasets, with our method consistently leading in all evaluated metrics.</p>
<p>Despite achieving satisfactory results in our experiments, there are still a couple of limitations: Computational Efficiency: The attention mechanism module may increase the computational complexity of the model while improving performance. Our future work will focus on further optimizing these modules to ensure improved computational efficiency while maintaining performance. Generality and Generalization: Although our method performed well on different datasets, its generality and generalization need to be strengthened. Future research will aim to widely validate the model&#x00027;s adaptability to various tasks and scenarios to ensure its robustness in practical applications. Future Outlook: Moving forward, we will continue in-depth research to further improve the attention mechanism module, exploring the integration of more advanced deep learning techniques. We will also investigate more data augmentation methods to enhance the model&#x00027;s adaptability to different data distributions. Ultimately, our goal is to develop a universal and efficient image generation model that provides viable solutions to real-world problems in the field of image processing.</p>
<p>Our research makes an important contribution to the field of jellyfish sting detection. First, we adopted a neural computing-based method, combining adaptive deep learning and the advanced YOLOv4 framework, to implement a high-precision jellyfish sting detection algorithm. This algorithm can quickly and accurately identify jellyfish stings in real-time scenarios, providing a reliable basis for timely treatment. Second, we construct a large-scale, diverse jellyfish sting dataset and accurately annotate it. This provides a basis for training and evaluating our algorithm, as well as a valuable resource for research and development in the field of jellyfish sting detection. Additionally, our study focused on practical applications of jellyfish sting detection. We apply our algorithm to real-time systems or applications to achieve continuous monitoring and detection of jellyfish stings, improving the efficiency and accuracy of jellyfish sting identification. These contributions will help promote the development of jellyfish sting detection technology, improve the efficiency and accuracy of jellyfish sting treatment, and protect public health and safety.</p></sec>
<sec sec-type="data-availability" id="s5">
<title>Data availability statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the article/supplementary material.</p></sec>
<sec sec-type="author-contributions" id="s6">
<title>Author contributions</title>
<p>CZ: Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Validation, Visualization, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing. HF: Conceptualization, Formal analysis, Investigation, Methodology, Project administration, Resources, Supervision, Visualization, Writing &#x02013; review &#x00026; editing. LX: Formal analysis, Funding acquisition, Investigation, Methodology, Project administration, Resources, Software, Supervision, Writing &#x02013; original draft, Writing &#x02013; review &#x00026; editing.</p></sec>
</body>
<back>
<sec sec-type="funding-information" id="s7">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec sec-type="disclaimer" id="s8">
<title>Publisher&#x00027;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Al-refai</surname> <given-names>G.</given-names></name> <name><surname>Al-refai</surname> <given-names>M.</given-names></name></person-group> (<year>2020</year>). <article-title>Road object detection using yolov3 and kitti dataset</article-title>. <source>Int. J. Adv. Comp. Sci. Appl</source>. <volume>11</volume>:<fpage>8</fpage>. <pub-id pub-id-type="doi">10.14569/IJACSA.2020.0110807</pub-id></citation>
</ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Su</surname> <given-names>D.</given-names></name></person-group> (<year>2019</year>). <article-title>Multi-modal fusion network with multi-scale multi-path and cross-modal interactions for rgb-d salient object detection</article-title>. <source>Pattern Recognit</source>. <volume>86</volume>, <fpage>376</fpage>&#x02013;<lpage>385</lpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2018.08.007</pub-id></citation>
</ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cunha</surname> <given-names>S. A.</given-names></name> <name><surname>Dinis-Oliveira</surname> <given-names>R. J.</given-names></name></person-group> (<year>2022</year>). <article-title>Raising awareness on the clinical and forensic aspects of jellyfish stings: a worldwide increasing threat</article-title>. <source>Int. J. Environ. Res. Public Health</source> <volume>19</volume>, <fpage>8430</fpage>. <pub-id pub-id-type="doi">10.3390/ijerph19148430</pub-id><pub-id pub-id-type="pmid">35886286</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dai</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name> <name><surname>Zhou</surname> <given-names>F.</given-names></name> <name><surname>Barnard</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>Attentional local contrast networks for infrared small target detection</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>59</volume>, <fpage>9813</fpage>&#x02013;<lpage>9824</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2020.3044958</pub-id></citation>
</ref>
<ref id="B5">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>C.</given-names></name> <name><surname>Cai</surname> <given-names>Q.</given-names></name> <name><surname>Ming</surname> <given-names>S.</given-names></name></person-group> (<year>2020</year>). <article-title>&#x0201C;Yolov4 object detection algorithm with efficient channel attention mechanism,&#x0201D;</article-title> in <source>2020 5th International Conference on Mechanical, Control and Computer Engineering (ICMCCE)</source> (<publisher-loc>Harbin</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1764</fpage>&#x02013;<lpage>1770</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://www.mdpi.com/1424-8220/21/23/8160">https://www.mdpi.com/1424-8220/21/23/8160</ext-link></citation>
</ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>M.</given-names></name> <name><surname>Bai</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Zhang</surname> <given-names>B.</given-names></name> <name><surname>Chang</surname> <given-names>Q.</given-names></name></person-group> (<year>2021</year>). <article-title>Real-time jellyfish classification and detection based on improved yolov3 algorithm</article-title>. <source>Sensors</source> <volume>21</volume>:<fpage>8160</fpage>. <pub-id pub-id-type="doi">10.3390/s21238160</pub-id><pub-id pub-id-type="pmid">34884161</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gao</surname> <given-names>W.</given-names></name> <name><surname>Liao</surname> <given-names>G.</given-names></name> <name><surname>Ma</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>G.</given-names></name> <name><surname>Liang</surname> <given-names>Y.</given-names></name> <name><surname>Lin</surname> <given-names>W.</given-names></name></person-group> (<year>2021</year>). <article-title>Unified information fusion network for multi-modal rgb-d and rgb-t salient object detection</article-title>. <source>IEEE Trans. Circuits Syst. Video Technol</source>. <volume>32</volume>, <fpage>2091</fpage>&#x02013;<lpage>2106</lpage>. <pub-id pub-id-type="doi">10.1109/TCSVT.2021.3082939</pub-id><pub-id pub-id-type="pmid">38043372</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>Z.</given-names></name> <name><surname>Lu</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Wu</surname> <given-names>R.</given-names></name> <name><surname>Huang</surname> <given-names>Z.</given-names></name></person-group> (<year>2022</year>). <article-title>Strategy to combine two functional components: efficient nano material development for iodine immobilization</article-title>. <source>Chemosphere</source> <volume>309</volume>:<fpage>136477</fpage>. <pub-id pub-id-type="doi">10.1016/j.chemosphere.2022.136477</pub-id><pub-id pub-id-type="pmid">36162517</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>R.</given-names></name> <name><surname>Pedoeem</surname> <given-names>J.</given-names></name> <name><surname>Chen</surname> <given-names>C.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Yolo-lite: a real-time object detection algorithm optimized for non-gpu computers,&#x0201D;</article-title> in <source>2018 IEEE International Conference on Big Data (big data)</source> (<publisher-loc>Seattle</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>2503</fpage>&#x02013;<lpage>2510</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Khamassi</surname> <given-names>M.</given-names></name> <name><surname>Mirolli</surname> <given-names>M.</given-names></name> <name><surname>Wallraven</surname> <given-names>C.</given-names></name></person-group> (<year>2023</year>). <article-title>Neurorobotics explores the human senses</article-title>. <source>Front. Neurorobot</source>. <volume>17</volume>:<fpage>1214871</fpage>. <pub-id pub-id-type="doi">10.3389/fnbot.2023.1214871</pub-id><pub-id pub-id-type="pmid">37283783</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>Y.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Lin</surname> <given-names>W.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name></person-group> (<year>2018</year>). <article-title>&#x0201C;Tiny-dsod: lightweight object detection for resource-restricted usages,&#x0201D;</article-title> in <source>arXiv</source>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/abs/1807.11013">https://arxiv.org/abs/1807.11013</ext-link></citation>
</ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>Z.</given-names></name> <name><surname>Xu</surname> <given-names>F.</given-names></name></person-group> (<year>2023</year>). <article-title>&#x0201C;Simulation of robot automatic control model based on artificial intelligence algorithm,&#x0201D;</article-title> in 2023 <italic>2nd International Conference on Artificial Intelligence and Autonomous Robot Systems (AIARS)</italic> (Bristol: IEEE), <fpage>535</fpage>&#x02013;<lpage>539</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Liu</surname> <given-names>X.</given-names></name> <name><surname>Zhang</surname> <given-names>B.</given-names></name></person-group> (<year>2023</year>). <article-title>Retinanet-vline: a flexible small target detection algorithm for efficient aggregation of information</article-title>. <source>Cluster Comput</source>. <volume>2023</volume>, <fpage>1</fpage>&#x02013;<lpage>13</lpage>. <pub-id pub-id-type="doi">10.1007/s10586-023-04109-4</pub-id></citation>
</ref>
<ref id="B14">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ma</surname> <given-names>Z.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Wu</surname> <given-names>L.</given-names></name> <name><surname>Zhang</surname> <given-names>L.</given-names></name> <name><surname>Zeng</surname> <given-names>Y.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>&#x0201C;Target detection and tracking of ground mobile robot based on improved single shot multibox detector network,&#x0201D;</article-title> in <source>2021 IEEE International Conference on Mechatronics and Automation (ICMA)</source> (<publisher-loc>Takamatsu</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>38</fpage>&#x02013;<lpage>43</lpage>.</citation>
</ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mahaur</surname> <given-names>B.</given-names></name> <name><surname>Mishra</surname> <given-names>K.</given-names></name> <name><surname>Kumar</surname> <given-names>A.</given-names></name></person-group> (<year>2023</year>). <article-title>An improved lightweight small object detection framework applied to real-time autonomous driving</article-title>. <source>Expert Syst. Appl</source>. <volume>234</volume>:<fpage>121036</fpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2023.121036</pub-id></citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Martin-Abadal</surname> <given-names>M.</given-names></name> <name><surname>Ruiz-Frau</surname> <given-names>A.</given-names></name> <name><surname>Hinz</surname> <given-names>H.</given-names></name> <name><surname>Gonzalez-Cid</surname> <given-names>Y.</given-names></name></person-group> (<year>2020</year>). <article-title>Jellytoring: real-time jellyfish monitoring based on deep learning object detection</article-title>. <source>Sensors</source> <volume>20</volume>:<fpage>1708</fpage>. <pub-id pub-id-type="doi">10.3390/s20061708</pub-id><pub-id pub-id-type="pmid">32204330</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nie</surname> <given-names>X.</given-names></name> <name><surname>Duan</surname> <given-names>M.</given-names></name> <name><surname>Ding</surname> <given-names>H.</given-names></name> <name><surname>Hu</surname> <given-names>B.</given-names></name> <name><surname>Wong</surname> <given-names>E. K.</given-names></name></person-group> (<year>2020</year>). <article-title>Attention mask r-cnn for ship detection and segmentation from remote sensing images</article-title>. <source>Ieee Access</source> <volume>8</volume>, <fpage>9325</fpage>&#x02013;<lpage>9334</lpage>. <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2964540</pub-id></citation>
</ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Obeso</surname> <given-names>A. M.</given-names></name> <name><surname>Benois-Pineau</surname> <given-names>J.</given-names></name> <name><surname>V&#x000E1;zquez</surname> <given-names>M. S. G.</given-names></name> <name><surname>Acosta</surname> <given-names>A.&#x000C1;. R</given-names></name></person-group>. (<year>2022</year>). <article-title>Visual vs internal attention mechanisms in deep neural networks for image classification and object detection</article-title>. <source>Pattern Recognit</source>. <volume>123</volume>:<fpage>108411</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2021.108411</pub-id></citation>
</ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname> <given-names>A. M.</given-names></name> <name><surname>Bose</surname> <given-names>R.</given-names></name> <name><surname>Bhaduri</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>A fast accurate fine-grain object detection model based on yolov4 deep neural network</article-title>. <source>Neural Comp. Appl</source>. <volume>2022</volume>, <fpage>1</fpage>&#x02013;<lpage>27</lpage>. <pub-id pub-id-type="doi">10.1007/s00521-021-06651-x</pub-id></citation>
</ref>
<ref id="B20">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sharma</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>&#x0201C;Information measure computation and its impact in mi coco dataset,&#x0201D;</article-title> in <source>2021 7th International Conference on Advanced Computing and Communication Systems (ICACCS)</source> (<publisher-loc>Coimbatore</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>1964</fpage>&#x02013;<lpage>1969</lpage>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>L.</given-names></name> <name><surname>Lv</surname> <given-names>X.</given-names></name> <name><surname>Lian</surname> <given-names>X.</given-names></name> <name><surname>Wang</surname> <given-names>G.</given-names></name></person-group> (<year>2021</year>). <article-title>Yolov4_drone: Uav image target detection based on an improved yolov4 algorithm</article-title>. <source>Comp. Electrical Eng</source>. <volume>93</volume>:<fpage>107261</fpage>. <pub-id pub-id-type="doi">10.1016/j.compeleceng.2021.107261</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tong</surname> <given-names>K.</given-names></name> <name><surname>Wu</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>Rethinking pascal-voc and ms-coco dataset for small object detection</article-title>. <source>J. Vis. Commun. Image Represent</source>. <volume>93</volume>:<fpage>103830</fpage>. <pub-id pub-id-type="doi">10.1016/j.jvcir.2023.103830</pub-id></citation>
</ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Veit</surname> <given-names>A.</given-names></name> <name><surname>Alldrin</surname> <given-names>N.</given-names></name> <name><surname>Chechik</surname> <given-names>G.</given-names></name> <name><surname>Krasin</surname> <given-names>I.</given-names></name> <name><surname>Gupta</surname> <given-names>A.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name></person-group> (<year>2017</year>). <article-title>&#x0201C;Learning from noisy large-scale datasets with minimal supervision,&#x0201D;</article-title> in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>, 839&#x02013;847. <pub-id pub-id-type="doi">10.1109/CVPR.2017.696</pub-id><pub-id pub-id-type="pmid">22331853</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vuong</surname> <given-names>D.-P.</given-names></name> <name><surname>Nguyen</surname> <given-names>T.-T.</given-names></name></person-group> (<year>2023</year>). <article-title>Fuzzy-proportional-integral-derivative-based controller for object tracking in mobile robots</article-title>. <source>Int. J. Elect. Comp. Eng</source>. <volume>13</volume>:<fpage>3</fpage>. <pub-id pub-id-type="doi">10.11591/ijece.v13i3.pp2498-2507</pub-id></citation>
</ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>K.</given-names></name> <name><surname>Liu</surname> <given-names>M.</given-names></name></person-group> (<year>2022</year>). <article-title>Toward structural learning and enhanced yolov4 network for object detection in optical remote sensing images</article-title>. <source>Adv. Theory Simulat</source>. <volume>5</volume>:<fpage>2200002</fpage>. <pub-id pub-id-type="doi">10.1002/adts.202200002</pub-id></citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>R.</given-names></name> <name><surname>Han</surname> <given-names>Z.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Cao</surname> <given-names>G.</given-names></name> <name><surname>Shen</surname> <given-names>T.</given-names></name> <name><surname>Cheng</surname> <given-names>X.</given-names></name> <name><surname>Tang</surname> <given-names>Y.</given-names></name></person-group> (<year>2021</year>). <article-title>Magnesium-functionalized ferro metal-carbon nanocomposite (mg-femec) for efficient uranium extraction from natural seawater</article-title>. <source>ACS ES&#x00026;T Water</source> <volume>1</volume>, <fpage>980</fpage>&#x02013;<lpage>990</lpage>. <pub-id pub-id-type="doi">10.1021/acsestwater.0c00262</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname> <given-names>J.</given-names></name> <name><surname>Sui</surname> <given-names>Z.</given-names></name> <name><surname>Xu</surname> <given-names>F.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name></person-group> (<year>2023</year>). <article-title>A novel model-free adaptive proportional-integral-derivative control method for speed-tracking systems of electric balanced forklifts</article-title>. <source>Appl. Sci</source>. <volume>13</volume>:<fpage>12816</fpage>. <pub-id pub-id-type="doi">10.3390/app132312816</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname> <given-names>C.</given-names></name> <name><surname>Yin</surname> <given-names>X.</given-names></name> <name><surname>Li</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>R.</given-names></name> <name><surname>Yu</surname> <given-names>H.</given-names></name> <name><surname>Xing</surname> <given-names>R.</given-names></name> <name><surname>Liu</surname> <given-names>S.</given-names></name> <name><surname>Li</surname> <given-names>P.</given-names></name></person-group> (<year>2024</year>). <article-title>Toxin metalloproteinases exert a dominant influence on pro-inflammatory response and anti-inflammatory regulation in jellyfish sting dermatitis</article-title>. <source>J. Proteomics</source> <volume>292</volume>:<fpage>105048</fpage>. <pub-id pub-id-type="doi">10.1016/j.jprot.2023.105048</pub-id><pub-id pub-id-type="pmid">37981009</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Yun</surname> <given-names>J.</given-names></name> <name><surname>Jiang</surname> <given-names>D.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Sun</surname> <given-names>Y.</given-names></name> <name><surname>Tao</surname> <given-names>B.</given-names></name> <name><surname>Kong</surname> <given-names>J.</given-names></name> <name><surname>Tian</surname> <given-names>J.</given-names></name> <name><surname>Tong</surname> <given-names>X.</given-names></name> <name><surname>Xu</surname> <given-names>M.</given-names></name> <name><surname>Fang</surname> <given-names>Z.</given-names></name></person-group> (<year>2022</year>). <article-title>Real-time target detection method based on lightweight convolutional neural network</article-title>. <source>Front. Bioeng. Biotechnol</source>. <volume>10</volume>:<fpage>861286</fpage>. <pub-id pub-id-type="doi">10.3389/fbioe.2022.861286</pub-id><pub-id pub-id-type="pmid">36051585</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zeng</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>B.</given-names></name> <name><surname>Zhu</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Underwater target detection based on faster r-cnn and adversarial occlusion network</article-title>. <source>Eng. Appl. Artif. Intell</source>. <volume>100</volume>:<fpage>104190</fpage>. <pub-id pub-id-type="doi">10.1016/j.engappai.2021.104190</pub-id></citation>
</ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Chen</surname> <given-names>N.</given-names></name> <name><surname>Chen</surname> <given-names>S.</given-names></name> <name><surname>Liu</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Novel fuzzy clustering algorithm with variable multi-pixel fitting spatial information for image segmentation</article-title>. <source>Pattern Recognit</source>. <volume>121</volume>:<fpage>108201</fpage>. <pub-id pub-id-type="doi">10.1016/j.patcog.2021.108201</pub-id></citation>
</ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>B.</given-names></name> <name><surname>Wang</surname> <given-names>C.</given-names></name> <name><surname>Fu</surname> <given-names>Q.</given-names></name> <name><surname>Han</surname> <given-names>Z.</given-names></name></person-group> (<year>2020</year>). <article-title>A novel pattern for infrared small target detection with generative adversarial network</article-title>. <source>IEEE Trans. Geosci. Remote Sens</source>. <volume>59</volume>:<fpage>4481</fpage>&#x02013;<lpage>4492</lpage>. <pub-id pub-id-type="doi">10.1109/TGRS.2020.3012981</pub-id></citation>
</ref>
</ref-list>
</back>
</article>