<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mar. Sci.</journal-id>
<journal-title>Frontiers in Marine Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mar. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-7745</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmars.2024.1470424</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Marine Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Detecting and quantifying deep sea benthic life using advanced object detection</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Iyer</surname>
<given-names>Karthik H.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2800390"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Marnor</surname>
<given-names>Camilla M.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn003">
<sup>&#x2020;</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2940353"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Schmid</surname>
<given-names>Daniel W.</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2941326"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Hartz</surname>
<given-names>Ebbe H.</given-names>
</name>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2940846"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Bergverk AS</institution>, <addr-line>Sandefjord</addr-line>, <country>Norway</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>AkerBP ASA</institution>, <addr-line>Lysaker</addr-line>, <country>Norway</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Elva G. Escobar-Briones, National Autonomous University of Mexico, Mexico</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Jianfeng Tong, Shanghai Ocean University, China</p>
<p>Annemiek Vink, Federal Institute For Geosciences and Natural Resources, Germany</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Karthik H. Iyer, <email xlink:href="mailto:karthik.iyer@bergwerk.com">karthik.iyer@bergwerk.com</email>
</p>
</fn>
<fn fn-type="other" id="fn003">
<p>&#x2020;ORCID: Camilla M. Marnor, <uri xlink:href="https://orcid.org/0009-0007-5584-5528">orcid.org/0009-0007-5584-5528</uri>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>13</day>
<month>01</month>
<year>2025</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>11</volume>
<elocation-id>1470424</elocation-id>
<history>
<date date-type="received">
<day>25</day>
<month>07</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>12</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2025 Iyer, Marnor, Schmid and Hartz</copyright-statement>
<copyright-year>2025</copyright-year>
<copyright-holder>Iyer, Marnor, Schmid and Hartz</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>We present a new dataset combined with the DeepSee model, which utilizes the YOLOv8 architecture, designed to rapidly and accurately detect benthic lifeforms in deep-sea environments of the North Atlantic. The dataset consists of 2,825 carefully curated images, encompassing 20,076 instances across 15 object-detection classes based on morphospecies from the phyla Arthropoda, Chordata, Cnidaria, Echinodermata, and Porifera. When benchmarked against a published dataset from the same region, DeepSee achieves high performance metrics, including an impressive mean Average Precision (mAP) score of 0.84, and produces very few false positives, ensuring reliable detection. The model processes images at 28&#x2013;50 frames per second (fps) for images sized at 1280 pixels, significantly increasing processing speed and reducing annotation workloads by over 1000 times when compared to manual annotation. While the model is not intended to replace the expertise of experienced biologists, it provides a valuable tool for accelerating data analysis and increasing efficiency. As additional data becomes available, augmenting the dataset and retraining the model will enable further improvements in detection capabilities. The dataset and model are designed for extensibility, allowing for the inclusion of other benthic lifeforms from the North Atlantic and beyond. This capability supports the creation of high-resolution maps of benthic life on the largely unexplored ocean floor of the Norwegian Continental Shelf (NCS) and other regions. This will facilitate informed decision-making in marine resource exploration, including mining operations, bottom trawling, and deep-sea pipeline laying, while also contributing to marine conservation and the sustainable management of deep-sea ecosystems.</p>
</abstract>
<kwd-group>
<kwd>deep sea benthic life</kwd>
<kwd>object detection</kwd>
<kwd>machine learning</kwd>
<kwd>marine resources</kwd>
<kwd>MPA (marine protected area)</kwd>
</kwd-group>
<counts>
<fig-count count="12"/>
<table-count count="3"/>
<equation-count count="0"/>
<ref-count count="45"/>
<page-count count="18"/>
<word-count count="6944"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Deep-Sea Environments and Ecology</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>The deep sea remains one of the least explored frontiers on Earth, with limited high-resolution data available on its biological diversity. Previous work in this domain has often been constrained by the low resolution and limited scope of available data (<xref ref-type="bibr" rid="B33">Ramirez-Llodra et&#xa0;al., 2010</xref>). Recent advancements in technology have resulted in large amounts of high-resolution images and video footage captured by remotely operated vehicles (ROVs). While this wealth of data holds immense potential for scientific discovery, e.g. of new megafaunal species or communities, the sheer volume makes manual annotation of observed megafaunal species impractical. This is where machine learning (ML) can play a pivotal role, automating the annotation process and enabling more efficient analysis of these vast datasets. Notably, studies (<xref ref-type="bibr" rid="B40">Schoening et&#xa0;al., 2016</xref>; <xref ref-type="bibr" rid="B23">Liu and Wang, 2021</xref>; <xref ref-type="bibr" rid="B9">Fu et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B24">Liu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B25">Lyu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B44">Xu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B11">Geisz et&#xa0;al., 2024</xref>) have demonstrated the efficacy of object detection/segmentation techniques in similar environments, underscoring the potential of ML in this field.</p>
<p>The objective here is to contribute to the emerging, high-resolution database on benthic life by leveraging ML to process and annotate the extensive video footage particularly in the North Atlantic. Object detection in computer vision involves identifying and localizing objects within an image. The method provides both the classification of objects and their precise locations within the image. This is particularly useful in various applications such as autonomous driving, security surveillance, and environmental monitoring. Over the past two decades, object detection has undergone a remarkable evolution, with several groundbreaking models significantly advancing the field. Region-based Convolutional Neural Networks (R-CNN) (<xref ref-type="bibr" rid="B13">Girshick et&#xa0;al., 2014</xref>) introduced a two-stage approach involving region proposals and classification. Fast R-CNN (<xref ref-type="bibr" rid="B12">Girshick, 2015</xref>) and Faster R-CNN (<xref ref-type="bibr" rid="B39">Ren et&#xa0;al., 2015</xref>) improved speed and efficiency by integrating these stages and introducing a Region Proposal Network. The Single Shot MultiBox Detector (SSD) (<xref ref-type="bibr" rid="B22">Liu et&#xa0;al., 2016</xref>) uses a single network to predict bounding boxes and class probabilities directly, enhancing speed while maintaining accuracy. The <italic>You Only Look Once</italic> (YOLO) (<xref ref-type="bibr" rid="B36">Redmon et&#xa0;al., 2016</xref>) family of models is a single stage detector that uses one pass of the network to identify and classify objects, significantly improving speed by predicting bounding boxes and class probabilities simultaneously. Subsequent versions (<xref ref-type="bibr" rid="B37">Redmon and Farhadi, 2017</xref>; <xref ref-type="bibr" rid="B41">Wang et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B38">Redmon and Farhadi, 2018</xref>; <xref ref-type="bibr" rid="B2">Bochkovskiy et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B16">Jocher, 2020</xref>; <xref ref-type="bibr" rid="B17">Jocher, 2023</xref>; <xref ref-type="bibr" rid="B42">Wang A. et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B43">Wang C.-Y. et&#xa0;al., 2024</xref>) introduced various enhancements including batch normalization, deeper networks, and optimized feature aggregation. These models&#xa0;offer various trade-offs between accuracy, speed, and computational&#xa0;complexity, with the choice depending on specific application requirements.</p>
<p>In this paper, we present the DeepSee dataset, a comprehensive collection of annotated images from the Arctic Mid-Ocean Ridge, the Norwegian Sea, and the Greenland Sea. This dataset is designed to support the development of ML models capable of detecting and classifying benthic organisms. The DeepSee object detection model, trained on this dataset, is capable of processing vast amounts of footage quickly with high precision and recall. The model provides a valuable addition to the traditional workflow of manual annotation by significantly reducing the load on human annotators. By deploying such models, we aim to streamline the annotation process, making it easier for biologists to conduct their research and ultimately supporting informed decision-making regarding deep-sea resource management and protection.</p>
<p>The implications of our work, and others like it, extend beyond mere academic inquiry; they are particularly relevant in the context of various anthropogenic activities that can disturb benthic ecosystems.</p>
<p>Mining Operations: As deep-sea mining becomes increasingly viable, particularly with recent proposals from the Norwegian government to open parts of its Exclusive Economic Zone (EEZ) for mining operations (<ext-link ext-link-type="uri" xlink:href="https://www.regjeringen.no/no/aktuelt/horing-av-forste-konsesjonsrunde-for-havbunnsmineraler/id3047008/">https://www.regjeringen.no/no/aktuelt/horing-av-forste-konsesjonsrunde-for-havbunnsmineraler/id3047008/</ext-link>) and the ongoing interest in mining polymetallic nodules in the Clarion-Clipperton Zone (CCZ) (<xref ref-type="bibr" rid="B14">Gollner et&#xa0;al., 2022</xref>), object detection models can help monitor biodiversity as well as community and abundance changes in these areas. By identifying sensitive habitats and taxa before mining activities commence, stakeholders can make informed decisions that minimize ecological impact.</p>
<p>Bottom Trawling: This fishing method has been criticized for its destructive effects on seafloor habitats over vast areas globally (<xref ref-type="bibr" rid="B18">Kroodsma et&#xa0;al., 2018</xref>). Our models can be employed to assess areas impacted by bottom trawling by detecting and cataloging benthic life before and after trawling events. This data can be used to guide regulations aimed at sustainable fishing practices.</p>
<p>Deep-Sea Pipeline Laying: The installation of pipelines for oil and gas transport poses risks to benthic ecosystems (<xref ref-type="bibr" rid="B6">Clare et&#xa0;al., 2023</xref>). By utilizing our object detection capabilities during pipeline construction projects, operators can identify critical habitats that need protection or monitoring during installation processes.</p>
<p>Environmental Monitoring: Continuous monitoring of deep-sea environments is crucial for assessing changes over time due to climate change or human activity. Our ML models can add to the toolbox of deep-sea benthic monitoring (<xref ref-type="bibr" rid="B21">Lim et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B10">Gallego et&#xa0;al., 2024</xref>) by automating the detection of shifts in species distribution or abundance, providing timely data that can influence conservation strategies.</p>
<p>Marine Protected Areas (MPAs): Effective management of MPAs requires robust data on biodiversity within these regions. Today, proposed MPAs of the NE Atlantic largely depend on biophysical habitat mapping founded on bathymetric analysis (<xref ref-type="bibr" rid="B8">Evans et&#xa0;al., 2015</xref>; <xref ref-type="bibr" rid="B20">Legrand et&#xa0;al., 2024</xref>). Our dataset and models can facilitate ongoing assessments of benthic life within MPAs, ensuring that these protected areas fulfill their conservation objectives.</p>
<p>Pollution Tracking: The detection of marine debris is increasingly important as pollution levels rise in oceanic environments. Our models can assist in identifying and quantifying debris impacts on benthic communities, contributing to efforts aimed at mitigating pollution effects.</p>
<p>In summary, integrating advanced machine learning techniques into deep-sea research workflows enhances our understanding of benthic ecosystems and provides valuable tools for addressing human impacts on these environments. Our work seeks to promote collaboration between technology and biology, contributing to informed decision-making in deep-sea resource management through accurate, rapid and comprehensive ecological data analysis.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Methods</title>
<sec id="s2_1">
<label>2.1</label>
<title>Model</title>
<p>In this study, we utilize the YOLOv8 model architecture as the basis of the DeepSee model for detecting benthic lifeforms on the ocean floor. The model is selected for its balance of speed and accuracy, making it well-suited for processing large volumes of underwater imagery efficiently. YOLOv8 begins by dividing the input image into a grid of cells. Each cell is responsible for detecting objects that fall within its area. The model uses a deep convolutional neural network (CNN) to extract important features from the image. This process identifies key details such as edges, shapes, and textures, which are crucial for recognizing different objects. For each cell in the grid, YOLOv8 predicts multiple bounding boxes. A bounding box is an imaginary rectangle that outlines where an object is located in the image. Along with the coordinates (center point, width, and height) of these boxes, the model also predicts a confidence score that indicates how likely it is that an object of a certain class (a particular morphospecies in this study) is present within that box. Each bounding box prediction comes with class probabilities. This means that for each box, YOLOv8 assigns a score to each possible object class, indicating how likely it is that the object belongs to each class. After predicting multiple bounding boxes for potential objects, YOLOv8 uses a technique called non-maxima suppression to eliminate overlapping boxes. This step ensures that only the most accurate bounding boxes are retained for each detected object, reducing clutter and improving clarity in the final output.</p>
<p>To detect benthic lifeforms, the DeepSee model is trained on a custom dataset comprising underwater images annotated with various benthic species. The training utilizes the YOLOv8x model weights (largest model in the family) as a starting point to leverage the advanced features of this architecture and to maximize prediction accuracy. Initially, the AdamW optimizer is employed for the first 10,000 iterations which helps with initial convergence. After 10,000 iterations, the optimizer is switched to Stochastic Gradient Descent (SGD) to refine the model further. SGD is known for its stability and convergence properties, which are beneficial in the later stages of training to fine-tune the weights precisely. To enhance the model&#x2019;s performance, data augmentation techniques are applied during training. These include random rotations, flips, scaling, color adjustments and image and label blending to simulate various underwater conditions and improve the model&#x2019;s robustness. The default loss function is used in training which is a composite of three components: &#x201c;box&#x201d;, &#x201c;dfl&#x201d;, and &#x201c;cls&#x201d;. The &#x201c;box&#x201d; loss refers to the loss associated with bounding box regression, ensuring the predicted boxes accurately localize the benthic organisms. The &#x201c;dfl&#x201d;, or Distribution Focal Loss, focuses on the distribution of predictions to enhance the model&#x2019;s confidence in detecting objects and minimizes the effects of class imbalance in the dataset. The &#x201c;cls&#x201d; component is the standard classification Cross Entropy Loss, which deals with the accuracy of class predictions. Other hyperparameters including the learning rate and loss function weighting are tuned on the dataset over multiple iterations, each lasting 200 epochs. Through this comprehensive training procedure, the DeepSee model is fine-tuned to effectively detect and classify benthic lifeforms. Final training is carried out over 400 epochs with an early stopping criterion of 100 epochs, i.e. the model will stop training if no improvement is detected over 100 epochs. The best model over these epochs is used for inference.</p>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Dataset</title>
<p>The DeepSee dataset is initially constructed using frame grabs (906 images) from videos captured by remotely operated vehicles (ROVs) during deep-sea surveys in the Arctic Mid-Ocean Ridge, the Norwegian Sea, and the Greenland Sea (<xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>). The locations are mostly from regions broadly classified as marine mountainous terrain. High-quality images, where instances of one or more classes are distinctly visible, are selected for the training dataset. Note that class in this context is a ML-term that represents a distinct object, not to be confused with the biological taxonomic rank &#x2018;class&#x2019;. An instance refers to an occurrence of a class in an image.</p>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Map of the Northeast Atlantic showing the locations of ROV footage used in the DeepSee dataset. The marine landscape layer is produced by the Geological Survey of Norway (<ext-link ext-link-type="uri" xlink:href="https://www.ngu.no/en/geologiske-kart/datasett">https://www.ngu.no/en/geologiske-kart/datasett</ext-link>).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g001.tif"/>
</fig>
<p>Variations in camera equipment, video resolution, ROV altitude and speed, and lighting conditions introduce significant diversity in image quality, presenting challenges in selecting appropriate training material. While this variation can enhance the model&#x2019;s ability to classify unseen data (<xref ref-type="bibr" rid="B11">Geisz et&#xa0;al., 2024</xref>), it is crucial to maintain high image and annotation quality to prevent an increase in false positives. Therefore, careful selection of training data is essential. The ROV camera perspectives range from top-down views to sub-parallel orientations relative to the seafloor, providing a comprehensive dataset with the organisms seen from multiple angles. Optimal video material includes segments with at least full HD resolution, sufficient lighting, minimal seafloor sampling activity to avoid obscured views from suspended particulate matter, and proximity to the seafloor to ensure good visibility of organisms and their morphological characteristics. Some images are cropped to exclude cases where organisms are not fully visible or cannot be clearly identified as a unique class to reduce erroneous detections. To enhance dataset diversity and increase the number of instances per class, additional images from published datasets have been incorporated. These include images of megabenthic communities from the Schulz Bank (<xref ref-type="bibr" rid="B29">Meyer et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B30">Meyer et&#xa0;al., 2023</xref>) (141 images), ophiuroids (<xref ref-type="bibr" rid="B27">Marnor, 2022</xref>) (74), and various fish and asteroids species from the Open Images Dataset V6/7 (<xref ref-type="bibr" rid="B19">Kuznetsova et&#xa0;al., 2020</xref>) (1778 images). It should be noted that the Open Images Dataset contains more than 10,000 images of fish and asteroids which largely consist of shallow water species. These images included many incorrect annotations that needed significant manual revision and curation. This comprehensive approach aims to improve the model&#x2019;s robustness and accuracy in detecting benthic life in the deep sea.</p>
<p>The resulting dataset comprises 2,825 images, each annotated with bounding boxes by a single observer. To ensure consistency, a second observer conducts a quality check of the annotations. Overall, the images contain 20,076 instances across 15 different lifeform classes (<xref ref-type="fig" rid="f2">
<bold>Figures&#xa0;2</bold>
</xref>, <xref ref-type="fig" rid="f3">
<bold>3</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S1</bold>
</xref>). The classes are chosen based on distinct morphospecies and groups of morphospecies that share a similar morphology. The classes must have enough instances in the video and image material so that they can be used for training, e.g. some organisms like the octopus <italic>Cirrotheutis muelleri</italic> have been observed in the video footage but cannot be used to train the model due to their rarity. Organism taxonomy is based on the World Register of Marine Species database. The observed deep-sea environment in the dataset ranges from sedimentary plains to hard substrate areas, and from desolate areas with few visible organisms to diversity hot spots like sponge grounds. Most of the detected classes are mainly associated with hard substrate. Some organisms, like sea stars, brittle stars, and sea anemones are observed both on hard and soft substrates. In sediment dominated areas single rocks or rocky outcrops are frequently observed with sponges, crinoids and/or soft corals.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Example instances of the different classes in the DeepSee dataset. The image of the Fish class is from the Open Image Dataset V6/7.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g002.tif"/>
</fig>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Distribution of instances and images per class in the DeepSee training/validation dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g003.tif"/>
</fig>
<p>
<bold>
<italic>Phylum: Arthropoda</italic>
</bold>
</p>
<list list-type="bullet">
<list-item>
<p>
<italic>Caridea</italic>: Several different morphospecies of shrimps. The deep-sea shrimp <italic>Bythocaris</italic> is the most abundant one in the training dataset. Only the body is annotated. Legs and antennas are not included in the bounding box as these features are usually not visible in the data.</p>
</list-item>
</list>
<p>
<bold>
<italic>Phylum: Chordata</italic>
</bold>
</p>
<list list-type="bullet">
<list-item>
<p>Chordata are divided into three different classes based on morphology: Fish, Fish <italic>&#x2013;</italic> deep sea, and Batoidea.</p>
</list-item>
<list-item>
<p>
<italic>Fish</italic>: Based on the downloaded images from Open Images and covers a large variety of fish, mostly tropical fish.</p>
</list-item>
<list-item>
<p>
<italic>Fish &#x2013; deep sea</italic>: Fish associated with the deep-sea seafloor in the ROV video material. Some relevant morphospecies in this class are Glacial eelpout (<italic>Lycodes frigidus</italic>), Arctic rockling (<italic>Gaidropsarus argentatus</italic>), Threadfin seasnail (<italic>Rhodichthys regina</italic>) and Black seasnail (<italic>Paraliparis bathybius</italic>) <italic>(</italic>
<xref ref-type="bibr" rid="B3">Brodnicke et&#xa0;al., 2023</xref>). Shortened to Fish-deep for annotation.</p>
</list-item>
<list-item>
<p>
<italic>Batoidea</italic>: Skates and rays. These can be morphologically distinguished from the other fish and are annotated as a separate class. Arctic skate (<italic>Amblyraja hyperborea</italic>) is an example of an observed morphospecies from this class.</p>
</list-item>
</list>    <p>
<bold>
<italic>Phylum: Cnidaria</italic>
</bold>
</p>
<list list-type="bullet">
<list-item>
<p>
<italic>Actiniaria</italic>: Instances of sea anemones. Actiniarians are observed both on hard substrates and in soft sediment areas. They are only annotated when the central disc and tentacles are visible. Individuals with withdrawn tentacles are not annotated since the morphology is not unique.</p>
</list-item>
<list-item>
<p>
<italic>Gersemia</italic>: Soft corals with morphology resembling that of genus <italic>Gersemia</italic>. An indicator of cauliflower coral gardens (<xref ref-type="bibr" rid="B1">Albrecht et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B35">Ramirez-Llodra et&#xa0;al., 2024</xref>).</p>
</list-item>
</list>    <p>
<bold>
<italic>Phylum: Echinodermata</italic>
</bold>
</p>
<list list-type="bullet">
<list-item>
<p>
<italic>Asteroidea</italic>: Instances of sea stars. All observations of sea stars, including individuals with more than 5 arms. Sea stars are observed on both hard and soft substrates, and on sponges.</p>
</list-item>
<list-item>
<p>
<italic>Ophiuroidea</italic>: Instances of brittle stars. Compared to sea stars, brittle stars have a rounder central disc and thinner arms that are more clearly separated from each other on the central disc. Brittle stars are observed in high abundances and densities in some areas, mainly soft sediment areas.</p>
</list-item>
<list-item>
<p>
<italic>Antedonidae</italic>: Instances of unstalked crinoids. They are usually observed on hard substrates and on other sponges, sometimes in high abundance. There are two relevant species that have a morphology too similar to be differentiated with ML currently: <italic>Poliometra prolixa</italic> and <italic>Heliometra glacialis (</italic>
<xref ref-type="bibr" rid="B34">Ramirez-Llodra et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B32">Pedersen et&#xa0;al., 2022</xref>). They are collectively annotated under the family Antedonidae. The instances are annotated when their characteristic morphology is visible either from the side or from above.</p>
</list-item>
</list>    <p>
<bold>
<italic>Phylum: Porifera</italic>
</bold>
</p>
<list list-type="bullet">
<list-item>
<p>
<italic>Lissodendoryx (Lissodendoryx) complicata</italic>: A relatively small, bush-shaped white demosponge that often can be identified based on its morphology (<xref ref-type="bibr" rid="B26">Marmen et&#xa0;al., 2019</xref>). <italic>L. (L.) complicata</italic> is a common sponge in arctic sponge grounds (<xref ref-type="bibr" rid="B28">Mayer and Piepenburg, 1996</xref>; <xref ref-type="bibr" rid="B31">Meyer et&#xa0;al., 2019</xref>). Shortened to Lissodendoryx for annotation.</p>
</list-item>
<list-item>
<p>
<italic>Asconema foliatum</italic>: Bush-shaped glass sponge like <italic>L. (L.) complicata</italic> but greyer and less dendritic. An indicator species for arctic sponge grounds (<xref ref-type="bibr" rid="B1">Albrecht et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B35">Ramirez-Llodra et&#xa0;al., 2024</xref>). Shortened to Asconema-fol for annotation.</p>
</list-item>
<list-item>
<p>
<italic>Asconema megaatrialia</italic>: Brownish, vase-shaped glass sponge. Shortened to Asconema-meg for annotation.</p>
</list-item>
<list-item>
<p>
<italic>Caulophacus (Caulophacus) arcticus</italic>: A common glass sponge in the Norwegian Sea and an indicator species for arctic sponge grounds (<xref ref-type="bibr" rid="B4">Buhl-Mortensen et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B1">Albrecht et&#xa0;al., 2020</xref>). C<italic>. (C.) arcticus</italic> is only annotated when the top and stalk are visible, as the top part alone can look like other sponges. Shortened to Caulophacus for annotation.</p>
</list-item>
<list-item>
<p>
<italic>Axinella/Phakellia</italic>: White fan-shaped sponges most likely belonging to the <italic>Phakellia</italic> and <italic>Axinella</italic> genera (<xref ref-type="bibr" rid="B4">Buhl-Mortensen et&#xa0;al., 2019</xref>; <xref ref-type="bibr" rid="B32">Pedersen et&#xa0;al., 2022</xref>). Indicators for hard-bottom sponge grounds (<xref ref-type="bibr" rid="B1">Albrecht et&#xa0;al., 2020</xref>).</p>
</list-item>
<list-item>
<p>
<italic>Rossellidae</italic>: Three species of white, vase-shaped glass sponges in the family Rosselidae grouped together due to similar morphology (<xref ref-type="bibr" rid="B30">Meyer et&#xa0;al., 2023</xref>): <italic>Schaudinnia rosea</italic>, <italic>Trichasterina borealis</italic> and <italic>Scyphidium septentrionale</italic>. These are structure-forming sponges and indicators for arctic sponge grounds (<xref ref-type="bibr" rid="B1">Albrecht et&#xa0;al., 2020</xref>; <xref ref-type="bibr" rid="B30">Meyer et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B35">Ramirez-Llodra et&#xa0;al., 2024</xref>).</p>
</list-item>
</list>
<p>Organisms are only annotated if it was possible to classify them into one of the classes without context (i.e., not by association with another individual from the same class present in the image). For some classes that tend to occur in clusters, like <italic>L. (L.) complicata</italic> and Rossellidae, it is challenging to distinguish individuals from each other. These are annotated as separate individuals where there is a clear spatial separation visible between the specimens in the image. <italic>L. (L.) complicata</italic> is only labelled if it is represented as at least a small &#x201c;bush&#x201d;. Small dendritic fractions are not annotated as they cannot be uniquely identified as such.</p>
</sec>
<sec id="s2_3">
<label>2.3</label>
<title>Training and validation</title>
<p>Training is carried out on the DeepSee dataset using a 75/25 split for training and validation over 400 epochs. The number of epochs is chosen based on multiple iterations of model training where a flattening of the performance metrics is observed after ca. 250-350 epochs (<xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>). The image size is set to 1280 pixels and batch size to 8 during training. We tested multiple batch sizes and found that increasing batch size results in a slight hit in model accuracy for our setup. Training is carried out on 4 Nvidia T4 GPUs. Object inference on a single image including post-processing takes ca. 35-60 milliseconds on a single Nvidia 4070 GPU, depending on the image complexity. The Yolov8 model uses the precision metric to identify the best model. See <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref> for the performance metrics. Precision is the fraction of relevant instances or true positives among all retrieved instances. Recall is the fraction of relevant instances (true positives) that were retrieved to the total number of ground truth instances. The F1 Score is the harmonic mean of precision and recall, providing a balanced assessment of a model&#x2019;s performance while considering both false positives and false negatives (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S1</bold>
</xref>). mAP50 is the mean average precision calculated at an intersection over union (IoU) threshold of 0.50 and is a measure of the model&#x2019;s accuracy. The trained model has a high mAP50 score of 0.84 over all classes. The high precision and mAP scores also show that false positives are kept to a minimum for most classes. This is also reflected in the confusion matrix where the detections are largely confined to the diagonal (<xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref>). A confusion matrix is a table layout that allows visualization of model performance by quantifying the detection errors or how the model confuses different classes during detection. For example, reading <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> vertically for the &#x2018;Fish&#x2019; class, the model correctly predicts the &#x2018;Fish&#x2019; class in 76 out of 100 cases and predicts it as &#x2018;Background&#x2019; in 23 out of 100 cases (false negative). When read horizontally, the model correctly predicts the &#x2018;Fish&#x2019; class in 76 out of 100 cases and incorrectly predicts instances of &#x2018;Batoidea&#x2019; as &#x2018;Fish&#x2019; in 19 out of 100 cases (false positive). The only classes with mAP50 scores lower than 0.8 are Fish-deep and Batoidea. The confusion matrix shows that these classes are often predicted as background, i.e. not detected, explaining the relatively low mAP50 scores. This is also reflected in low recall scores for these classes. This is most likely due to the low number of images and instances of these classes in the training dataset. The mAP50-95 score, or mean Average Precision over IoU thresholds from 0.5 to 0.95, measures model performance at multiple higher levels of localization precision, making it a more rigorous metric than mAP50. mAP50-95 scores are generally lower than mAP50 scores as it assesses not only detection accuracy but also the tightness of bounding boxes.</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Training performance metrics.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g004.tif"/>
</fig>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>Performance metrics of the trained model for all detected classes in the validation data.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Class</th>
<th valign="middle" align="center">Instances (GT)</th>
<th valign="middle" align="center">Instances (Detected)</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">mAP50</th>
<th valign="middle" align="center">mAP50-95</th>
<th valign="middle" align="center">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">All</td>
<td valign="top" align="center">3960</td>
<td valign="top" align="center">3873</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.66</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr>
<td valign="bottom" align="center">Fish</td>
<td valign="top" align="center">900</td>
<td valign="top" align="center">779</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.69</td>
<td valign="top" align="center">0.81</td>
</tr>
<tr>
<td valign="bottom" align="center">Fish-deep</td>
<td valign="top" align="center">17</td>
<td valign="top" align="center">9</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.47</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.63</td>
<td valign="top" align="center">0.62</td>
</tr>
<tr>
<td valign="bottom" align="center">Batoidea</td>
<td valign="top" align="center">16</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.50</td>
<td valign="top" align="center">0.58</td>
<td valign="top" align="center">0.51</td>
<td valign="top" align="center">0.57</td>
</tr>
<tr>
<td valign="bottom" align="center">Caridea</td>
<td valign="top" align="center">128</td>
<td valign="top" align="center">113</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.56</td>
<td valign="top" align="center">0.82</td>
</tr>
<tr>
<td valign="bottom" align="center">Asteroidea</td>
<td valign="top" align="center">106</td>
<td valign="top" align="center">106</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.91</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.87</td>
</tr>
<tr>
<td valign="bottom" align="center">Ophiuroidea</td>
<td valign="top" align="center">248</td>
<td valign="top" align="center">222</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.86</td>
</tr>
<tr>
<td valign="bottom" align="center">Actiniaria</td>
<td valign="top" align="center">277</td>
<td valign="top" align="center">297</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.85</td>
</tr>
<tr>
<td valign="bottom" align="center">Antedonidae</td>
<td valign="top" align="center">734</td>
<td valign="top" align="center">754</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.90</td>
<td valign="top" align="center">0.59</td>
<td valign="top" align="center">0.86</td>
</tr>
<tr>
<td valign="top" align="center">Gersemia</td>
<td valign="top" align="center">55</td>
<td valign="top" align="center">52</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.53</td>
<td valign="top" align="center">0.84</td>
</tr>
<tr>
<td valign="top" align="center">Axinella/Phakellia</td>
<td valign="top" align="center">291</td>
<td valign="top" align="center">263</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.68</td>
<td valign="top" align="center">0.79</td>
</tr>
<tr>
<td valign="bottom" align="center">Lissodendoryx</td>
<td valign="top" align="center">946</td>
<td valign="top" align="center">1025</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.81</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.57</td>
<td valign="top" align="center">0.79</td>
</tr>
<tr>
<td valign="bottom" align="center">Asconema-fol</td>
<td valign="top" align="center">29</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.7</td>
<td valign="top" align="center">0.79</td>
</tr>
<tr>
<td valign="bottom" align="center">Caulophacus</td>
<td valign="top" align="center">24</td>
<td valign="top" align="center">23</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.85</td>
</tr>
<tr>
<td valign="bottom" align="center">Asconema-meg</td>
<td valign="top" align="center">37</td>
<td valign="top" align="center">32</td>
<td valign="top" align="center">1.00</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.92</td>
</tr>
<tr>
<td valign="bottom" align="center">Rossellidae</td>
<td valign="top" align="center">152</td>
<td valign="top" align="center">154</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.8</td>
<td valign="top" align="center">0.86</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>GT, Ground Truth.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Confusion matrix for the trained model on DeepSee validation data.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g005.tif"/>
</fig>
<p>The dataset is also utilized to train other models from the YOLO family for comparative analysis and to verify the stability of dataset performance (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S2</bold>
</xref>). The tests indicate that the performance of the DeepSee dataset remains consistent across different generations of YOLO models and model sizes. Inference time for a 1280x1280 pixel image ranges from ca. 20 to 35 milliseconds per frame. Testing with smaller models resulted in only a minor decrease in performance metrics from 0.84 to 0.81 mAP. This is promising, as these smaller models can be deployed on systems with limited computational power, such as remotely operated vehicles (ROVs), and where near real-time decision-making may be key.</p>
</sec>
</sec>
<sec id="s3" sec-type="results">
<label>3</label>
<title>Results and discussion</title>
<p>The trained DeepSee model achieves a high mAP50 score of 0.84 on validation data, indicating strong accuracy and minimal false positives across most classes. Inference times for images sized at 1280 pixels range from approximately 20 to 35 milliseconds per frame, demonstrating the model&#x2019;s efficiency. The DeepSee model is further benchmarked against 48 images selected from the Meyer (<xref ref-type="bibr" rid="B29">Meyer et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B30">Meyer et&#xa0;al., 2023</xref>) dataset covering the range of habitat types and organisms present in the DeepSee dataset. Note that these images are not used for DeepSee model training and validation. The Meyer dataset catalogues benthic organisms on the Schulz Bank, a seamount of the Artic Mid-Ocean Ridge, using ROV video footage from 580 to 2700m depth. The Meyer study also provides instance counts of the identified organisms for each image in the dataset. However, these counts include not only clearly visible organisms but also obscured or fragmented organisms that do not display the characteristic features of their class as required for ML classification. These additional counts were identified by tentative association with other organisms in the image or by physical sampling (Heidi Kristina Meyer, pers. comm.). For example, in <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S2</bold>
</xref>, where all clearly visible organisms in the DeepSee dataset are annotated, 3 instances of Lissodendoryx are identifiable. In contrast, the Meyer study counts 7 Lissodendoryx instances here (<xref ref-type="bibr" rid="B29">Meyer et&#xa0;al., 2022</xref>). Therefore, for benchmarking and comparative purposes in this study, the dataset has been manually re-annotated to align with the classification and methodology used for the DeepSee model as outlined in Section 2 to provide the ground truth instance count, rather than using the instance counts provided by the Meyer study. Inference is carried out using the DeepSee model and the results are compared to the annotations. The resulting validation metrics are shown in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> and the confusion matrix in <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>. The validation metrics and confusion matrix show that the model does an excellent job in detecting most organisms in the dataset with mAP50 scores higher than 0.7. The only exception to this is the Ophiuroidea class where the metrics are relatively low. False positives, with respect to other classes in the dataset, are minimized except for the Fish-deep class (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref> is an example that shows a comparison between the manually annotated (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7A</bold>
</xref>) and DeepSee detected (<xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7B</bold>
</xref>) images. The Caridea, Antedonidae and Lissodendoryx classes with high mAP50 scores are annotated/detected in the example figure. The example figure has only 1 instance of Caridea annotated but the model returns two detected instances. As it stands, the precision and recall of this class would be 0.5 and 1, respectively. However, the &#x2018;false positive&#x2019; detection is indeed a missed manual ground truth annotation of the Caridea class. Accounting for this (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S3</bold>
</xref>), the precision and recall of this class reaches a perfect score of 1 for this class in the example (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>). This effect is also observed in the other classes present in the figure. There are 15 Antedonidae instances detected by the model of which 4 are &#x2018;false positive&#x2019;. In reality, there are no false positives when the manual annotations are corrected. The precision and recall for this class increase from 0.73 and 0.79, respectively, to 1 and 0.83. Similarly, 15 instances of Lissodendoryx are detected with 1 &#x2018;false-positive&#x2019; which is a misannotated instance. When the annotations are corrected, the precision and recall of this class are both 1.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Performance metrics of the trained model for all detected classes in the Meyer dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Class</th>
<th valign="middle" align="center">Instances (GT)</th>
<th valign="middle" align="center">Instances (Detected)</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">mAP50</th>
<th valign="middle" align="center">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">All</td>
<td valign="top" align="center">906</td>
<td valign="top" align="center">881</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.69</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.75</td>
</tr>
<tr>
<td valign="bottom" align="center">Fish-deep</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">1.00</td>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.80</td>
</tr>
<tr>
<td valign="bottom" align="center">Caridea</td>
<td valign="top" align="center">6</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">0.76</td>
<td valign="top" align="center">0.83</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.80</td>
</tr>
<tr>
<td valign="bottom" align="center">Asteroidea</td>
<td valign="top" align="center">21</td>
<td valign="top" align="center">17</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.82</td>
</tr>
<tr>
<td valign="bottom" align="center">Ophiuroidea</td>
<td valign="top" align="center">11</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">0.27</td>
<td valign="top" align="center">0.32</td>
<td valign="top" align="center">0.30</td>
</tr>
<tr>
<td valign="bottom" align="center">Actiniaria</td>
<td valign="top" align="center">326</td>
<td valign="top" align="center">335</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.83</td>
</tr>
<tr>
<td valign="bottom" align="center">Antedonidae</td>
<td valign="top" align="center">31</td>
<td valign="top" align="center">34</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.84</td>
</tr>
<tr>
<td valign="bottom" align="center">Gersemia</td>
<td valign="top" align="center">46</td>
<td valign="top" align="center">36</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.57</td>
<td valign="top" align="center">0.70</td>
<td valign="top" align="center">0.65</td>
</tr>
<tr>
<td valign="bottom" align="center">Lissodendoryx</td>
<td valign="top" align="center">328</td>
<td valign="top" align="center">294</td>
<td valign="top" align="center">0.85</td>
<td valign="top" align="center">0.67</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.75</td>
</tr>
<tr>
<td valign="bottom" align="center">Asconema-fol</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1.00</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.86</td>
</tr>
<tr>
<td valign="bottom" align="center">Rossellidae</td>
<td valign="top" align="center">127</td>
<td valign="top" align="center">139</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.78</td>
<td valign="top" align="center">0.80</td>
<td valign="top" align="center">0.78</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>GT, Ground Truth.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Confusion matrix for the trained model on the Meyer dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g006.tif"/>
</fig>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Figure showing the comparison between the manually annotated <bold>(A)</bold> and detected labels <bold>(B)</bold> in an example image. Instances that are labelled in the ground truth image and not found by the DeepSee model, and vice-versa, are encircled in black. The average precision and recall of the model detections for the image are 0.72 and 0.83, respectively. However, it is evident that the &#x2018;false positives&#x2019; detected by the model are indeed valid detections that were missed in the annotated dataset. When corrected, the average precision and recall increase to 1 and 0.91, respectively. Images are from the Meyer dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g007.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Performance metrics of the trained model for all detected classes in the revised Meyer dataset.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="middle" align="center">Class</th>
<th valign="middle" align="center">Instances (GT)</th>
<th valign="middle" align="center">Instances (Detected)</th>
<th valign="middle" align="center">Precision</th>
<th valign="middle" align="center">Recall</th>
<th valign="middle" align="center">mAP50</th>
<th valign="middle" align="center">F1</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="bottom" align="center">All</td>
<td valign="top" align="center">1012</td>
<td valign="top" align="center">881</td>
<td valign="top" align="center">0.93</td>
<td valign="top" align="center">0.74</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.82</td>
</tr>
<tr>
<td valign="bottom" align="center">Fish-deep</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">1.00</td>
<td valign="top" align="center">0.57</td>
<td valign="top" align="center">0.79</td>
<td valign="top" align="center">0.73</td>
</tr>
<tr>
<td valign="bottom" align="center">Caridea</td>
<td valign="top" align="center">8</td>
<td valign="top" align="center">7</td>
<td valign="top" align="center">1.00</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.94</td>
<td valign="top" align="center">0.93</td>
</tr>
<tr>
<td valign="bottom" align="center">Asteroidea</td>
<td valign="top" align="center">22</td>
<td valign="top" align="center">17</td>
<td valign="top" align="center">1.00</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.89</td>
<td valign="top" align="center">0.87</td>
</tr>
<tr>
<td valign="bottom" align="center">Ophiuroidea</td>
<td valign="top" align="center">18</td>
<td valign="top" align="center">12</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.50</td>
<td valign="top" align="center">0.61</td>
<td valign="top" align="center">0.60</td>
</tr>
<tr>
<td valign="bottom" align="center">Actiniaria</td>
<td valign="top" align="center">373</td>
<td valign="top" align="center">335</td>
<td valign="top" align="center">0.95</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.90</td>
</tr>
<tr>
<td valign="bottom" align="center">Antedonidae</td>
<td valign="top" align="center">38</td>
<td valign="top" align="center">34</td>
<td valign="top" align="center">0.97</td>
<td valign="top" align="center">0.87</td>
<td valign="top" align="center">0.92</td>
<td valign="top" align="center">0.92</td>
</tr>
<tr>
<td valign="bottom" align="center">Gersemia</td>
<td valign="top" align="center">50</td>
<td valign="top" align="center">36</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.62</td>
<td valign="top" align="center">0.77</td>
<td valign="top" align="center">0.72</td>
</tr>
<tr>
<td valign="bottom" align="center">Lissodendoryx</td>
<td valign="top" align="center">346</td>
<td valign="top" align="center">294</td>
<td valign="top" align="center">0.86</td>
<td valign="top" align="center">0.73</td>
<td valign="top" align="center">0.82</td>
<td valign="top" align="center">0.79</td>
</tr>
<tr>
<td valign="bottom" align="center">Asconema-fol</td>
<td valign="top" align="center">4</td>
<td valign="top" align="center">3</td>
<td valign="top" align="center">1.00</td>
<td valign="top" align="center">0.75</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.86</td>
</tr>
<tr>
<td valign="bottom" align="center">Rossellidae</td>
<td valign="top" align="center">146</td>
<td valign="top" align="center">139</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.84</td>
<td valign="top" align="center">0.88</td>
<td valign="top" align="center">0.86</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>GT, Ground Truth.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>The Fish-deep class has a perfect precision value of 1 but a relatively low recall value of 0.67 (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>) which tells us that although there are no false positives for this class, the class is either not detected or has been mislabeled as Batoidea or Rossellidae (<xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref>). The mislabeling of the Fish-deep class can be attributed to the features of the labelled classes in the image (<xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>). In <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8A</bold>
</xref>, the upper part of the organism is perceived as an angular wedge due to the proximity of the fins to the upper body. This feature is also present in the Batoidea class and is possibly the cause of the mislabel. In <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8B</bold>
</xref>, the Fish-deep instance in the top center of the image is cut off at the image edge. Consequently, the instance looks somewhat like class Rossellidae (see other Rossellidae instances in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>) and is mislabeled as such, albeit with a low confidence value.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>
<bold>(A)</bold> and <bold>(B)</bold> show the instances where the class Fish-deep has been mislabeled as Batoidea and Rossellidae, respectively. <bold>(C)</bold> shows the unlabeled image in <bold>(B)</bold> to emphasize the mislabeled Rossellidae instance (black ellipses). Images are from the Meyer dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g008.tif"/>
</fig>
<p>The Gersemia class also has a relatively low recall value of 0.57 (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). The low recall value is because several Gersemia instances are simply not detected (background) (<xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>) and is most likely due to the limited number of instances in the training dataset. This is also true for other detected classes but occurs less frequently due to better training data with respect to these classes, e.g. Lissodendoryx and Rossellidae (<xref ref-type="fig" rid="f9">
<bold>Figures&#xa0;9</bold>
</xref>, <xref ref-type="fig" rid="f10">
<bold>10</bold>
</xref>).</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>Example showing manually annotated <bold>(A)</bold> and model detected <bold>(B)</bold> labels for a selected image illustrating the low recall value of the Gersemia class (soft corals). Instances that are labelled in the ground truth image and not found by the DeepSee model, and vice-versa, are encircled in black. Images are from the Meyer dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g009.tif"/>
</fig>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Example showing manually annotated <bold>(A)</bold> and model detected <bold>(B)</bold> labels for a selected image illustrating the low metric scores of the Ophiuroidea class. Instances that are labelled in the ground truth image and not found by the DeepSee model, and vice-versa, are encircled in black. Images are from the Meyer dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g010.tif"/>
</fig>
<p>The Ophiuroidea show very low scores for all metrics throughout the dataset (<xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref>). The cause for these low scores is shown exemplarily in <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref>. There are 3 instances of the class annotated in the image while there are only 2 instances detected by model. Moreover, the detections do not correspond to the annotations, i.e. the model failed to detect the annotated Ophiuroidea and instead found additional, valid instances that were not annotated. The reason for this is twofold. Firstly, Ophiuroidea instances in all images in the dataset are relatively small and the model architecture has problems detecting very small objects. Secondly, in the dataset, these instances are always present on spicule mats that contain a lot of white detritus with sizes and shapes similar to the Ophiuroidea which makes model detection problematic. Many instances are so obscured and blurry that they can only be manually identified with context, i.e. they look only somewhat similar but can be identified as such due to the presence of other brittlestars in the image. The precision and recall of the model for this image without annotation corrections are both 0. When corrected, the precision and recall values increase to 1 and 0.4, respectively.</p>
<p>The validation dataset can then be corrected to account for the instances that are detected by the DeepSee model but were not annotated in the dataset, i.e. manual labelling was conservative, or the organism was simply missed during manual annotation. Note that the instances for which the model does not detect the organism are not corrected. This increases the metrics (<xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>, <xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref>) of the Caridea and Ophiuroidea classes where the model correctly identified the previously unlabeled organism in the images. The high precision values across the board show that the DeepSee model is very robust when it comes to correctly identifying benthic life while keeping false positives to a minimum. In most cases, the recall scores are also relatively high with some exceptions due to factors such as inadequate training data and feature similarity between classes.</p>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Confusion matrix for the trained model on the revised Meyer dataset.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g011.tif"/>
</fig>
</sec>
<sec id="s4">
<label>4</label>
<title>Comparison to other models</title>
<p>Recently, other models have also been developed to detect benthic life, achieving varying degrees of success in terms of accuracy and reliability. These studies (<xref ref-type="bibr" rid="B9">Fu et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B45">Zhang et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B25">Lyu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B44">Xu et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B5">Cai et&#xa0;al., 2024</xref>) focus on refining model architecture to improve detection in complex marine environments with highly variable scenes and increase model efficiency. The models achieve high mAP scores (between 0.7 and 0.85) that are largely in the same range as the DeepSee model. Interestingly, a recent model (<xref ref-type="bibr" rid="B7">Cuvelier et&#xa0;al., 2024</xref>) trained on a dataset from the Clarion-Clipperton Zone (CCZ) in the NE Pacific demonstrates high recall values (~0.90) but very low precision (~0.13) despite using relatively well-lit images captured by a camera consistently positioned approximately 1.5 meters above the seafloor and facing orthogonal to the seafloor. Although this model could be potentially used as a general object detector, the high rate of false positives (~86%) would necessitate substantial manual correction efforts to ensure accuracy.</p>
<p>Direct comparison of the DeepSee model with other such object detection models is challenging due to several factors. Firstly, the training dataset used in these studies is different, which directly influences performance metrics and applicability. More importantly, many of the models and/or their trained weights are not publicly available, limiting the ability to replicate results or conduct thorough evaluations. We compare DeepSee to the MBARI/FathomNet model that has been also developed to detect marine benthos and where the weights and models are publicly available (<ext-link ext-link-type="uri" xlink:href="https://huggingface.co/FathomNet/MBARI-315k-yolov8">https://huggingface.co/FathomNet/MBARI-315k-yolov8</ext-link>). It should be noted that the released weights are almost a year old and may be outdated. To be fair, DeepSee classes that are not comparable to the ones found in the MBARI model, and vice-versa, are ignored when computing performance metrics.</p>
<p>FathomNet is an impressive, open-source image database that has been specifically built to train AI models to help detect marine life. Like DeepSee, the MBARI model also uses YOLOv8x (model size is inferred from model parameter count) as the base object detection model and is trained on a FathomNet dataset containing 499 classes. Currently, the FathomNet dataset contains more than 110,000 images and 303,000 localizations. However, the number of images and instances in the dataset used for training the model has not been provided. Inference is performed on the revised Meyer dataset using the available model weights. Detections using the MBARI model classes have been relabeled to the appropriate DeepSee class for comparison (<xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Figure S4</bold>
</xref>, <xref ref-type="supplementary-material" rid="SM1">
<bold>Supplementary Table S3</bold>
</xref>). In some cases, the MBARI model identifies the same DeepSee class as different classes. All such classes are relabeled. For example, the DeepSee class &#x2018;Rossellidae&#x2019; is detected as &#x2018;Porifera&#x2019; and &#x2018;Hexactinellida&#x2019; by the MBARI model and are all relabeled to &#x2018;Rossellidae&#x2019;. The &#x2018;Lissodendoryx&#x2019; and &#x2018;Asconema-fol&#x2019; DeepSee classes which are present in the revised Meyer dataset are not detected by the MBARI model and are not used when computing performance metrics. Validation of the MBARI model on the revised Meyer dataset results in precision, recall and mAP scores of 0.73, 0.15 and 0.25, respectively. While the MBARI model demonstrates reasonable performance in correctly identifying organisms, as indicated by its fair precision score, it struggles to detect all organisms that are clearly visible in the images, resulting in very poor recall scores. The DeepSee model and dataset significantly outperform the MBARI model (<xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>).</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Comparison of detections between the DeepSee <bold>(A)</bold> and MBARI <bold>(B)</bold> models.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fmars-11-1470424-g012.tif"/>
</fig>
<p>The DeepSee model metrics show that ML methods can successfully automate benthic lifeform detection with relatively high accuracy thereby reducing the manual annotation load significantly. Manual annotation of an hour long ROV video where 1 frame is annotated per second would take ca. 50 hours to annotate, assuming each frame contains 10 instances with each instance requiring 5 seconds. The DeepSee model would, for the same video, require only 126-180 seconds, which is more than a thousand-fold increase in speed. It should be noted that to do this, the imaged lifeforms need to have clearly visible and defined morphologies or feature sets that can be usefully extracted, learned and subsequently detected by the model. The model cannot contextually label lifeforms as an experienced biologist would be able to do. This kind of reasoning and association is currently beyond the capabilities of models such as presented here and would require a trained professional to correctly identify these instances. The methodology presented here can help create high-resolution maps of benthic life in the deep sea by combining location metadata from the ROV with model object detections. This approach has the potential to overcome the typical data sparseness associated with deep-sea lifeform mapping. Traditionally, due to limited data availability, such maps are produced with granularity ranging from hundreds to tens of thousands of square kilometers (<xref ref-type="bibr" rid="B20">Legrand et&#xa0;al., 2024</xref>; <xref ref-type="bibr" rid="B35">Ramirez-Llodra et&#xa0;al., 2024</xref>), which can lead to an incomplete or inaccurate assessment of target regions. With the techniques presented in this study, map resolutions down to the square meter scale or even lower (<xref ref-type="bibr" rid="B15">Hartz et&#xa0;al., 2024</xref>) can be achieved, offering a much more detailed understanding of species and biomass distribution. Such precision could transform current perceptions of deep-sea biodiversity and community composition, and profoundly impact decision-making processes related to resource management and conservation in deep-sea environments.</p>
</sec>
<sec id="s5" sec-type="conclusions">
<label>5</label>
<title>Conclusions</title>
<p>This study demonstrates that the state-of-the-art DeepSee object detection model, trained on a carefully curated dataset that focuses on the Arctic Ocean Ridge, the Norwegian Sea and the Greenland Sea, can effectively detect benthic lifeforms in challenging deep-sea environments characterized by variation in lighting, perspective, object occlusion and imaging equipment. We show that the model can successfully localize organisms with high precision and achieve high mAP scores when sufficient training data is available. Lower recall scores are obtained in only a few cases and can be directly attributed to a lack of training data and/or the presence of highly obscured, small instances in the validation data, such as ophiuroids in spicule mats. The model metrics also inspire confidence in detection certainty with minimal false-positive detections.</p>
<p>Thus, DeepSee forms a strong foundation for future annotation tasks, integral to the workflow of marine scientists and surveyors in the North Atlantic. Although a valuable tool, the model cannot replace the expertise of scientists and <italic>in-situ</italic> sampling. The model annotations should be subject to expert verification and additional annotations will be required for cases where the model misses instances due to contextual or environmental challenges. Nevertheless, model deployment serves as a valuable foundation and will significantly reduce the time and workload needed to annotate such images by many orders of magnitude. This, in turn, will accelerate the creation of high-resolution maps of the seafloor, enhancing our understanding of the distribution of life in these regions and aiding in making quantitative, informed resource allocation decisions.</p>
<p>Future work will focus on expanding the DeepSee dataset by incorporating new detection classes to enhance its versatility and applicability. Additionally, efforts will be made to explore the identification of benthic organisms in regions beyond the North Atlantic, thereby broadening the dataset&#x2019;s geographical scope. This expansion will not only improve the model&#x2019;s robustness but also contribute to a more comprehensive understanding of benthic ecosystems globally. Furthermore, modifications to model architecture to enhance performance while maintaining computational efficiency is also worth investigating. Potential adjustments may include refining convolutional layers, integrating attention mechanisms, and experimenting with alternative loss functions. These improvements could yield valuable advancements in detection capabilities, particularly in the complex environments of deep-sea imagery, representing a promising direction for ongoing research.</p>
</sec>
</body>
<back>
<sec id="s6" sec-type="data-availability">
<title>Data availability statement</title>
<p>The raw data supporting the conclusions of this article will be made available by the authors, without undue reservation.</p>
</sec>
<sec id="s7" sec-type="author-contributions">
<title>Author contributions</title>
<p>KI: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. CM: Conceptualization, Data curation, Investigation, Visualization, Writing &#x2013; original draft. DS: Conceptualization, Investigation, Methodology, Project administration, Resources, Writing &#x2013; review &amp; editing. EH: Conceptualization, Project administration, Resources, Writing &#x2013; review &amp; editing.</p>
</sec>
<sec id="s8" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare that no financial support was received for the research, authorship, and/or publication of this article.</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We would like to thank AkerBP giving permission to publish this study and the Norwegian Offshore Directorate, University of Bergen and Adepth Minerals for supplying ROV videos for model training. We would also like to thank Heidi Kristina Meyer for an insightful discussion about the Schulz Bank study. We thank Jianfeng Tong and Annemiek Vink for their constructive reviews that helped strengthen the manuscript.  ChatGPT-4o by OpenAI was used to refine the manuscript.</p>
</ack>
<sec id="s9" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Authors KI, CM, and DS were employed by company Bergverk AS. Author EH was employed by company AkerBP ASA.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s10" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec id="s11" sec-type="supplementary-material">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fmars.2024.1470424/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fmars.2024.1470424/full#supplementary-material</ext-link>
</p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Albrecht</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Beazley</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Braga-Henriques</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Cardenas</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Carreiro-Silva</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Cola&#xe7;o</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>ICES/NAFO joint working group on deep-water ecology (WGDEC)</article-title>. doi:&#xa0;<pub-id pub-id-type="doi">10.17895/ices.pub.6095</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Yolov4: Optimal speed and accuracy of object detection</article-title>. <source>arXiv preprint arXiv:2004.10934</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2004.10934</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Brodnicke</surname> <given-names>O. B.</given-names>
</name>
<name>
<surname>Meyer</surname> <given-names>H. K.</given-names>
</name>
<name>
<surname>Busch</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Xavier</surname> <given-names>J. R.</given-names>
</name>
<name>
<surname>Knudsen</surname> <given-names>S. W.</given-names>
</name>
<name>
<surname>M&#xf8;ller</surname> <given-names>P. R.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Deep-sea sponge derived environmental DNA analysis reveals demersal fish biodiversity of a remote Arctic ecosystem</article-title>. <source>Environ. DNA</source> <volume>5</volume>, <fpage>1405</fpage>&#x2013;<lpage>1417</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/edn3.v5.6</pub-id>
</citation>
</ref>
<ref id="B4">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Buhl-Mortensen</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Burgos</surname> <given-names>J. M.</given-names>
</name>
<name>
<surname>Steingrund</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Buhl-Mortensen</surname> <given-names>P.</given-names>
</name>
<name>
<surname>&#xd3;lafsd&#xf3;ttir</surname> <given-names>S. H.</given-names>
</name>
<name>
<surname>Ragnarsson</surname> <given-names>S. &#xc1;.</given-names>
</name>
</person-group> (<year>2019</year>). <source>Vulnerable marine ecosystems (VMEs): Coral and sponge VMEs in Arctic and sub-Arctic waters&#x2013;Distribution and threats</source> Vol. <volume>2019519</volume> (<publisher-loc>Denmark</publisher-loc>: <publisher-name>Nordic Council of Ministers</publisher-name>).</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cai</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Mo</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A Lightweight underwater detector enhanced by Attention mechanism, GSConv and WIoU on YOLOv8</article-title>. <source>Sci. Rep.</source> <volume>14</volume>, <fpage>25797</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-024-75809-z</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Clare</surname> <given-names>M. A.</given-names>
</name>
<name>
<surname>Lichtschlag</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Paradis</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Barlow</surname> <given-names>N. L. M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>Assessing the impact of the global subsea telecommunications network on sedimentary organic carbon stocks</article-title>. <source>Nat. Commun.</source> <volume>14</volume>, <fpage>2080</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41467-023-37854-6</pub-id>
</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cuvelier</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Zurowietz</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Nattkemper</surname> <given-names>T. W.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Deep learning&#x2013;assisted biodiversity assessment in deep-sea benthic megafauna communities: a case study in the context of polymetallic nodule mining</article-title>. <source>Front. Mar. Sci.</source> <volume>11</volume>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmars.2024.1366078</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Evans</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Peckett</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Howell</surname> <given-names>K. L.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Combined application of biophysical habitat mapping and systematic conservation planning to assess efficiency and representativeness of the existing High Seas MPA network in the Northeast Atlantic</article-title>. <source>ICES J. Mar. Sci.</source> <volume>72</volume>, <fpage>1483</fpage>&#x2013;<lpage>1497</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1093/icesjms/fsv012</pub-id>
</citation>
</ref>
<ref id="B9">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Fu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A case study of utilizing YOLOT based quantitative detection algorithm for marine benthos</article-title>. <source>Ecol. Inf.</source> <volume>70</volume>, <fpage>101603</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2022.101603</pub-id>
</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gallego</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Arias</surname> <given-names>M. B.</given-names>
</name>
<name>
<surname>Corral-Lou</surname> <given-names>A.</given-names>
</name>
<name>
<surname>D&#xed;ez-Vives</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Neave</surname> <given-names>E. F.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>C.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>North Atlantic deep-sea benthic biodiversity unveiled through sponge natural sampler DNA</article-title>. <source>Commun. Biol.</source> <volume>7</volume>, <fpage>1015</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s42003-024-06695-4</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Geisz</surname> <given-names>J. K.</given-names>
</name>
<name>
<surname>Wernette</surname> <given-names>P. A.</given-names>
</name>
<name>
<surname>Esselman</surname> <given-names>P. C.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Classification of lakebed geologic substrate in autonomously collected benthic imagery using machine learning</article-title>. <source>Remote Sens.</source> <volume>16</volume>, <fpage>1264</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs16071264</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2015</year>). <source>Proceedings of the IEEE international conference on computer vision</source>, <publisher-loc>Santiago, Chile</publisher-loc>. <fpage>1440</fpage>&#x2013;<lpage>1448</lpage>.</citation>
</ref>
<ref id="B13">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Donahue</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Darrell</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Malik</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2014</year>). <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>, <publisher-loc>Columbus, Ohio</publisher-loc>. <fpage>580</fpage>&#x2013;<lpage>587</lpage>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gollner</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Haeckel</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Janssen</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Lefaible</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Molari</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Papadopoulou</surname> <given-names>S.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Restoration experiments in polymetallic nodule areas</article-title>. <source>Integr. Environ. Assess. Manag.</source> <volume>18</volume>, <fpage>682</fpage>&#x2013;<lpage>696</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1002/ieam.4541</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Hartz</surname> <given-names>E. H.</given-names>
</name>
<name>
<surname>Iyer</surname> <given-names>K. H.</given-names>
</name>
<name>
<surname>Marnor</surname> <given-names>C. M.</given-names>
</name>
<name>
<surname>Schmid</surname> <given-names>D. W.</given-names>
</name>
</person-group> (<year>2024</year>). <source>DeepSee (v0.1.0)</source>. Github Repository. <uri xlink:href="https://github.com/Aker-BP-Open-Research/DeepSee">https://github.com/Aker-BP-Open-Research/DeepSee</uri>
<uri xlink:href="https://github.com/Aker-BP-Open-Research/DeepSee">https://github.com/Aker-BP-Open-Research/DeepSee</uri>.</citation>
</ref>
<ref id="B16">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Jocher</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2020</year>). <source>YOLOv5 by Ultralytics</source>. Available online at: <uri xlink:href="https://github.com/ultralytics/yolov5">https://github.com/ultralytics/yolov5</uri>.</citation>
</ref>
<ref id="B17">
<citation citation-type="web">
<person-group person-group-type="author">
<name>
<surname>Jocher</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2023</year>). <source>YOLOv8 by Ultralytics</source>. Available online at: <uri xlink:href="https://github.com/ultralytics/ultralytics">https://github.com/ultralytics/ultralytics</uri>.</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kroodsma</surname> <given-names>D. A.</given-names>
</name>
<name>
<surname>Mayorga</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Hochberg</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Miller</surname> <given-names>N. A.</given-names>
</name>
<name>
<surname>Boerder</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Ferretti</surname> <given-names>F.</given-names>
</name>
<etal/>
</person-group>. (<year>2018</year>). <article-title>Tracking the global footprint of fisheries</article-title>. <source>Science</source> <volume>359</volume>, <fpage>904</fpage>&#x2013;<lpage>908</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1126/science.aao5646</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kuznetsova</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>The open images dataset v4: Unified image classification, object detection, and visual relationship detection at scale</article-title>. <source>Int. J. Comput. Vision</source> <volume>128</volume>, <fpage>1956</fpage>&#x2013;<lpage>1981</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11263-020-01316-z</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Legrand</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Boulard</surname> <given-names>M.</given-names>
</name>
<name>
<surname>O&#x2019;Connor</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Kutti</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>Identifying priorities for the protection of deep-sea species and habitats in the Nordic Seas</article-title>.</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lim</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wheeler</surname> <given-names>A. J.</given-names>
</name>
<name>
<surname>Price</surname> <given-names>D. M.</given-names>
</name>
<name>
<surname>O&#x2019;Reilly</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Harris</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Conti</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Influence of benthic currents on cold-water coral habitats: a combined benthic monitoring and 3D photogrammetric investigation</article-title>. <source>Sci. Rep.</source> <volume>10</volume>, <fpage>19433</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1038/s41598-020-76446-y</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Anguelov</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Erhan</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Szegedy</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Reed</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Fu</surname> <given-names>C.-Y.</given-names>
</name>
<etal/>
</person-group>. (<year>2016</year>). <source>Computer Vision&#x2013;ECCV 2016: 14th European Conference, Amsterdam, The Netherlands</source> (<publisher-name>Springer</publisher-name>), <fpage>21</fpage>&#x2013;<lpage>37</lpage>. <italic>Proceedings, Part I 14</italic>.</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A quantitative detection algorithm based on improved faster R-CNN for marine benthos</article-title>. <source>Ecol. Inf.</source> <volume>61</volume>, <elocation-id>101228</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2021.101228</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
</person-group> (<year>2023</year>). <source>Artificial Intelligence Oceanography</source>. Eds. <person-group person-group-type="editor">
<name>
<surname>Li</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>F.</given-names>
</name>
</person-group> (<publisher-loc>Singapore</publisher-loc>: <publisher-name>Springer Nature Singapore</publisher-name>), <fpage>323</fpage>&#x2013;<lpage>346</lpage>.</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Lyu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Xu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>EFP-YOLO: A quantitative detection algorithm for marine benthic organisms</article-title>. <source>Ocean Coast. Manage.</source> <volume>243</volume>, <fpage>106770</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ocecoaman.2023.106770</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Marmen</surname> <given-names>M. B.</given-names>
</name>
<name>
<surname>Tompkins</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Harrington</surname> <given-names>N.</given-names>
</name>
<name>
<surname>Savard-Drouin</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Wells</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Baker</surname> <given-names>E.</given-names>
</name>
<etal/>
</person-group>. (<year>2019</year>). <article-title>Sponges from the 2010-2014 paamiut multispecies trawl surveys, eastern arctic and subarctic: class demospongiae, subclass heteroscleromorpha, order poecilosclerida, families microcionidae, acarnidae and esperiopsidae</article-title>. <source>Fish. Oceans Canada</source>.</citation>
</ref>
<ref id="B27">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Marnor</surname> <given-names>C. M.</given-names>
</name>
</person-group> (<year>2022</year>). <source>Mapping distribution patterns of brittle stars using ROV-based imaging</source> (<publisher-loc>Norway</publisher-loc>: <publisher-name>NTNU</publisher-name>).</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Mayer</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Piepenburg</surname> <given-names>D.</given-names>
</name>
</person-group> (<year>1996</year>). <article-title>Epibenthic community patterns on the continental slope off East Greenland at 75* N</article-title>. <source>Mar. Ecol. Prog. Ser.</source> <volume>143</volume>, <fpage>151</fpage>&#x2013;<lpage>164</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3354/meps143151</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname> <given-names>H. K.</given-names>
</name>
<name>
<surname>Davies</surname> <given-names>A. J.</given-names>
</name>
<name>
<surname>Roberts</surname> <given-names>E. M.</given-names>
</name>
<name>
<surname>Xavier</surname> <given-names>J. R.</given-names>
</name>
<name>
<surname>Ribeiro</surname> <given-names>P. A.</given-names>
</name>
<name>
<surname>Glenner</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <source>Megafauna abundance records during 2017 and 2018 SponGES Cruises (GSGS2017110 and GS2018108) with RV G.O. Sars and ROV &#xc6;gir 6000 on Schulz Bank, Arctic Mid-Ocean Ridge <italic>PANGAEA</italic>
</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.1594/PANGAEA.949920</pub-id>
</citation>
</ref>
<ref id="B30">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname> <given-names>H. K.</given-names>
</name>
<name>
<surname>Davies</surname> <given-names>A. J.</given-names>
</name>
<name>
<surname>Roberts</surname> <given-names>E. M.</given-names>
</name>
<name>
<surname>Xavier</surname> <given-names>J. R.</given-names>
</name>
<name>
<surname>Ribeiro</surname> <given-names>P. A.</given-names>
</name>
<name>
<surname>Glenner</surname> <given-names>H.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Beyond the tip of the seamount: Distinct megabenthic communities found beyond the charismatic summit sponge ground on an arctic seamount (Schulz Bank, Arctic Mid-Ocean Ridge)</article-title>. <source>Deep Sea Res. Part I: Oceanogr. Res. Papers</source> <volume>191</volume>, <elocation-id>103920</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.dsr.2022.103920</pub-id>
</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Meyer</surname> <given-names>H. K.</given-names>
</name>
<name>
<surname>Roberts</surname> <given-names>E. M.</given-names>
</name>
<name>
<surname>Rapp</surname> <given-names>H. T.</given-names>
</name>
<name>
<surname>Davies</surname> <given-names>A. J.</given-names>
</name>
</person-group> (<year>2019</year>). <article-title>Spatial patterns of arctic sponge ground fauna and demersal fish are detectable in autonomous underwater vehicle (AUV) imagery</article-title>. <source>Deep Sea Res. Part I: Oceanogr. Res. Papers</source> <volume>153</volume>, <fpage>103137</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.dsr.2019.103137</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Pedersen</surname> <given-names>R. B.</given-names>
</name>
<name>
<surname>Rydland Olsen</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Barreyre</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Bjerga</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Denny</surname> <given-names>A.</given-names>
</name>
<etal/>
</person-group>. (<year>2022</year>). <article-title>Fagutredning mineralressurser i norskehavet landskapstrekk, naturtyper og bentiske &#xf8;kosystemer</article-title>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramirez-Llodra</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Brandt</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Danovaro</surname> <given-names>R.</given-names>
</name>
<name>
<surname>De Mol</surname> <given-names>B.</given-names>
</name>
<name>
<surname>Escobar</surname> <given-names>E.</given-names>
</name>
<name>
<surname>German</surname> <given-names>C. R.</given-names>
</name>
<etal/>
</person-group>. (<year>2010</year>). <article-title>Deep, diverse and definitely different: unique attributes of the world&#x2019;s largest ecosystem</article-title>. <source>Biogeosciences</source> <volume>7</volume>, <fpage>2851</fpage>&#x2013;<lpage>2899</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.5194/bg-7-2851-2010</pub-id>
</citation>
</ref>
<ref id="B34">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramirez-Llodra</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Hilario</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Paulsen</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Costa</surname> <given-names>C. V.</given-names>
</name>
<name>
<surname>Bakken</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Johnsen</surname> <given-names>G.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Benthic communities on the Mohn&#x2019;s treasure mound: implications for management of seabed mining in the Arctic mid-ocean ridge</article-title>. <source>Front. Mar. Sci.</source> <volume>7</volume>, <elocation-id>490</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.3389/fmars.2020.00490</pub-id>
</citation>
</ref>
<ref id="B35">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ramirez-Llodra</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Meyer</surname> <given-names>H. K.</given-names>
</name>
<name>
<surname>Bluhm</surname> <given-names>B. A.</given-names>
</name>
<name>
<surname>Brix</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Brandt</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Dannheim</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>The emerging picture of a diverse deep Arctic Ocean seafloor: From habitats to ecosystems</article-title>. <source>Elementa: Sci. Anthropocene</source> <volume>12</volume>, <fpage>00140</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1525/elementa.2023.00140</pub-id>
</citation>
</ref>
<ref id="B36">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Divvala</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>You only look once: Unified, real-time object detection</article-title> <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>. <publisher-loc>Las Vegas, USA</publisher-loc>. <fpage>779</fpage>&#x2013;<lpage>788</lpage>.</citation>
</ref>
<ref id="B37">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2017</year>). <source>Proceedings of the IEEE conference on computer vision and pattern recognition</source>. <publisher-loc>Honolulu, USA</publisher-loc>.  <fpage>7263</fpage>&#x2013;<lpage>7271</lpage>.</citation>
</ref>
<ref id="B38">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Redmon</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Farhadi</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2018</year>). <article-title>Yolov3: An incremental improvement</article-title>. <source>arXiv preprint arXiv:1804.02767</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1804.02767</pub-id>
</citation>
</ref>
<ref id="B39">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Ren</surname> <given-names>S.</given-names>
</name>
<name>
<surname>He</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Girshick</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Sun</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2015</year>). <article-title>Faster r-cnn: Towards real-time object detection with region proposal networks</article-title>. <source>IEEE transactions on pattern analysis and machine intelligence</source> <volume>39</volume> (<issue>6</issue>), <fpage>1137</fpage>-<lpage>1149</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.1506.01497</pub-id>
</citation>
</ref>
<ref id="B40">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schoening</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Kuhn</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Jones</surname> <given-names>D. O. B.</given-names>
</name>
<name>
<surname>Simon-Lledo</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Nattkemper</surname> <given-names>T. W.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Fully automated image segmentation for benthic resource assessment of poly-metallic nodules</article-title>. <source>Methods Oceanogr.</source> <volume>15-16</volume>, <fpage>78</fpage>&#x2013;<lpage>89</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.mio.2016.04.002</pub-id>
</citation>
</ref>
<ref id="B41">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Bochkovskiy</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>YOLOv7: Trainable bag-of-freebies sets new state-of-the-art for real-time object detectors</article-title>. <source>Proceedings of the IEEE/CVF conference on computer vision and pattern recognition</source>. <publisher-loc>Nashville, USA</publisher-loc>, <fpage>7464</fpage>&#x2013;<lpage>7475</lpage>.</citation>
</ref>
<ref id="B42">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Chen</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Lin</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Han</surname> <given-names>J.</given-names>
</name>
<etal/>
</person-group>. (<year>2024</year>). <article-title>YOLOv10: real-time end-to-end object detection</article-title>. <source>arXiv preprint arXiv:2405.14458</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2405.14458</pub-id>
</citation>
</ref>
<ref id="B43">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>C.-Y.</given-names>
</name>
<name>
<surname>Yeh</surname> <given-names>I.-H.</given-names>
</name>
<name>
<surname>Liao</surname> <given-names>H.-Y. M.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>YOLOv9: learning what you want to learn using programmable gradient information</article-title>. <source>arXiv preprint arXiv:2402.13616</source>. doi:&#xa0;<pub-id pub-id-type="doi">10.48550/arXiv.2402.13616</pub-id>
</citation>
</ref>
<ref id="B44">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Xu</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Liu</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Lyu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Yan</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>MAD-YOLO: A quantitative detection algorithm for dense small-scale marine benthos</article-title>. <source>Ecol. Inf.</source> <volume>75</volume>, <fpage>102022</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2023.102022</pub-id>
</citation>
</ref>
<ref id="B45">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Yongpan</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Xianchong</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Yong</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Lyu</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>Q.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>YoloXT: A object detection algorithm for marine benthos</article-title>. <source>Ecol. Inf.</source> <volume>72</volume>, <elocation-id>101923</elocation-id>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2022.101923</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>