<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Aquac.</journal-id>
<journal-title>Frontiers in Aquaculture</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Aquac.</abbrev-journal-title>
<issn pub-type="epub">2813-5334</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/faquc.2024.1365123</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Aquaculture</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Forecasting ocean hypoxia in salmonid fish farms</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name>
<surname>Cerqueira</surname>
<given-names>Vitor</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="author-notes" rid="fn001">
<sup>*</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2616876"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Pimentel</surname>
<given-names>Jo&#xe3;o</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2746717"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Korus</surname>
<given-names>Jennie</given-names>
</name>
<xref ref-type="aff" rid="aff3">
<sup>3</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2624309"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bravo</surname>
<given-names>Francisco</given-names>
</name>
<xref ref-type="aff" rid="aff4">
<sup>4</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2621338"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Amorim</surname>
<given-names>Joana</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<xref ref-type="aff" rid="aff2">
<sup>2</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Oliveira</surname>
<given-names>Mariana</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/2624967"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Swanson</surname>
<given-names>Andrew</given-names>
</name>
<xref ref-type="aff" rid="aff5">
<sup>5</sup>
</xref>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Filgueira</surname>
<given-names>Ram&#xf3;n</given-names>
</name>
<xref ref-type="aff" rid="aff6">
<sup>6</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/379701"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Grant</surname>
<given-names>Jon</given-names>
</name>
<xref ref-type="aff" rid="aff7">
<sup>7</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/792243"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Torgo</surname>
<given-names>Luis</given-names>
</name>
<xref ref-type="aff" rid="aff1">
<sup>1</sup>
</xref>
<uri xlink:href="https://loop.frontiersin.org/people/527767"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/project-administration/"/>
<role content-type="https://credit.niso.org/contributor-roles/resources/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
</contrib>
</contrib-group>
<aff id="aff1">
<sup>1</sup>
<institution>Faculty of Computer Science, Dalhousie University</institution>, <addr-line>Halifax, NS</addr-line>, <country>Canada</country>
</aff>
<aff id="aff2">
<sup>2</sup>
<institution>Vector Institute for Artificial Intelligence</institution>, <addr-line>Toronto, ON</addr-line>, <country>Canada</country>
</aff>
<aff id="aff3">
<sup>3</sup>
<institution>Innovasea Marine Systems Canada</institution>, <addr-line>Bedford, NS</addr-line>, <country>Canada</country>
</aff>
<aff id="aff4">
<sup>4</sup>
<institution>Ocean and Atmosphere Department, Fundaci&#xf3;n CSIRO Chile Research</institution>, <addr-line>Santiago</addr-line>, <country>Chile</country>
</aff>
<aff id="aff5">
<sup>5</sup>
<institution>Cooke Aquaculture Inc.</institution>, <addr-line>Saint John, NB</addr-line>, <country>Canada</country>
</aff>
<aff id="aff6">
<sup>6</sup>
<institution>Marine Affairs Program, Dalhousie University</institution>, <addr-line>Halifax, NS</addr-line>, <country>Canada</country>
</aff>
<aff id="aff7">
<sup>7</sup>
<institution>Department of Oceanography, Dalhousie University</institution>, <addr-line>Halifax, NS</addr-line>, <country>Canada</country>
</aff>
<author-notes>
<fn fn-type="edited-by">
<p>Edited by: Prabhugouda Siriyappagouder, Nord University, Norway</p>
</fn>
<fn fn-type="edited-by">
<p>Reviewed by: Clara Cordeiro, University of Algarve, Portugal</p>
<p>Jinran Wu, Australian Catholic University, Australia</p>
</fn>
<fn fn-type="corresp" id="fn001">
<p>*Correspondence: Vitor Cerqueira, <email xlink:href="mailto:vcerqueira@fe.up.pt">vcerqueira@fe.up.pt</email>
</p>
</fn>
</author-notes>
<pub-date pub-type="epub">
<day>09</day>
<month>07</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>3</volume>
<elocation-id>1365123</elocation-id>
<history>
<date date-type="received">
<day>03</day>
<month>01</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>18</day>
<month>06</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#xa9; 2024 Cerqueira, Pimentel, Korus, Bravo, Amorim, Oliveira, Swanson, Filgueira, Grant and Torgo</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Cerqueira, Pimentel, Korus, Bravo, Amorim, Oliveira, Swanson, Filgueira, Grant and Torgo</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<sec>
<title>Introduction</title>
<p>Hypoxia is defined as a critically low-oxygen condition of water, which, if prolonged, can be harmful to fish and many other aquatic species. In the context of ocean salmon fish farming, early detection of hypoxia events is critical for farm managers to mitigate these events to reduce fish stress, however in complex natural systems accurate forecasting tools are limited. The goal of this research is to use a machine learning approach to forecast oxygen concentration and predict hypoxia events in marine net-pen salmon farms.</p>
</sec>
<sec>
<title>Methods</title>
<p>The developed model is based on gradient boosting and works in two stages. First, we apply auto-regression to build a forecasting model that predicts oxygen concentration levels within a cage. We take a global forecasting approach by building a model using the historical data provided by sensors at several marine fish farms located in eastern Canada. Then, the forecasts are transformed into binary probabilities that indicate the likelihood of a low-oxygen event. We leverage the cumulative distribution function to compute these probabilities.</p>
</sec>
<sec>
<title>Results and discussion</title>
<p>We tested our model in a case study that included several cages across 14 fish farms. The experiments suggest that the model can detect future hypoxic events with a commercially acceptable false alarm rate. The resulting probabilistic predictions and oxygen concentration forecasts can help salmon farmers to prioritize resources, and reduce harm to crops.</p>
</sec>
</abstract>
<kwd-group>
<kwd>fish farms</kwd>
<kwd>oxygen dynamics</kwd>
<kwd>hypoxia</kwd>
<kwd>time series forecasting</kwd>
<kwd>exceedance probability</kwd>
</kwd-group>
<counts>
<fig-count count="15"/>
<table-count count="4"/>
<equation-count count="5"/>
<ref-count count="32"/>
<page-count count="15"/>
<word-count count="6937"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-in-acceptance</meta-name>
<meta-value>Disease and Health Management</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec id="s1" sec-type="intro">
<label>1</label>
<title>Introduction</title>
<p>Fish farms produce large quantities of marine protein efficiently, reducing pressure on wild fish populations. They promote biodiversity and thus, the sustainability of wild fishing operations (<xref ref-type="bibr" rid="B23">Subasinghe et&#xa0;al., 2009</xref>). Overall, fish farms are an increasingly important food production system in our society for both economic and environmental reasons, e.g. reducing overfishing.</p>
<p>Good farm management and fish health, critically require reduction of fish stress and the ability to detect poor water quality. In this regard, one of the main challenges for ocean farms is maintaining optimal levels of dissolved oxygen in cages. Variable farm conditions such as high water temperatures, algae blooms, and or high stocking densities can all exasperate oxygen availability, potentially triggering a low-oxygen event (<xref ref-type="bibr" rid="B30">Yokoyama, 1997</xref>; <xref ref-type="bibr" rid="B5">Burke et&#xa0;al., 2021</xref>). These events, also referred to as hypoxia episodes, occur when the oxygen level in the water falls below 7<italic>mgL<sup>&#x2212;</sup>
</italic>
<sup>1</sup> (<xref ref-type="bibr" rid="B18">Oppedal et&#xa0;al., 2011</xref>), though the severity depends on the specific farm environment, age of salmon, and the available reactive mitigation tools (i.e., oxygen dispersal systems) (<xref ref-type="bibr" rid="B20">Remen, 2012</xref>).</p>
<p>Hypoxia events cause stress to fish, negatively impacting fish welfare and productivity, and in the most severe cases can lead to mortalities, causing significant economic loss to producers. As such, the availability of more accurate hypoxia prediction tools would both improve the well-being of fish and the economic management of farms. Similarly, this tool could help farmers reduce significant costs associated with mitigation technologies.</p>
<p>The goal of this work is to build a model to i) forecast oxygen concentration and ii) predict hypoxia events on farms using machine learning. Specifically, at each time step, we aim to predict the probability that the oxygen level will fall below a critical threshold. A probabilistic approach is useful to decision-makers within farms. For example, if there is a high likelihood of a low-oxygen event, a farmer can proactively turn on mitigation systems rather than waiting for the low-oxygen event to occur, which can cause stress to fish, and delay return to more optimal oxygen levels.</p>
<p>The probabilistic forecasting of binary events is usually tackled as a classification problem. However, a classifier is unable to predict the future values of oxygen &#x2013; only the likelihood of a hypoxic event. Complementing event probability estimates with oxygen forecasts is valuable for having both an efficient means to alert operators of impending low dissolved oxygen events and the ability to analyze temporal trends in dissolved oxygen. In effect, we apply an auto-regressive approach to forecast oxygen dynamics. Then, we resort to the cumulative distribution function to convert these forecasts into a hypoxic event probability following <xref ref-type="bibr" rid="B6">Cerqueira and Torgo (2022)</xref>.</p>
<p>The developed model was validated using data from 14 fish farms located in different locations across eastern Canada. Each farm contains several cages where fish are grown in variable stocking densities, leading to different micro-climates for each cage. The results of the experiments showed that the model was able to accurately predict critical events at the cage level on each of the farms. Overall, this case study demonstrated the effectiveness of using machine learning to forecast critical events in ocean fish farms.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Case study</title>
<p>In this section, we present the case study used in this work. We start by describing the data structure and its main characteristics (Section 2.1). We also carry out an exploratory data analysis to uncover valuable insights before building a forecasting model (Section 2.2).</p>
<sec id="s2_1">
<label>2.1</label>
<title>Data description</title>
<p>We can define a database of <inline-formula>
<mml:math display="inline" id="im1">
<mml:mi>n</mml:mi>
</mml:math>
</inline-formula> fish farms as <inline-formula>
<mml:math display="inline" id="im2">
<mml:mrow>
<mml:mi>&#x2131;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mi>n</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. Each fish farm <inline-formula>
<mml:math display="inline" id="im3">
<mml:mrow>
<mml:mi>F</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>&#x2131;</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> contains several cages <inline-formula>
<mml:math display="inline" id="im4">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>F</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>m</mml:mi>
<mml:mi>F</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <italic>m<sub>F</sub>
</italic> is the number of cages in the farm. Finally, each cage <inline-formula>
<mml:math display="inline" id="im5">
<mml:mrow>
<mml:mi>C</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>F</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> can be represented as a time series <inline-formula>
<mml:math display="inline" id="im6">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>C</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im7">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> is the observation of cage <inline-formula>
<mml:math display="inline" id="im8">
<mml:mi>C</mml:mi>
</mml:math>
</inline-formula> at time <italic>i</italic> and <inline-formula>
<mml:math display="inline" id="im9">
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the number of observations in the cage. Each observation <inline-formula>
<mml:math display="inline" id="im10">
<mml:mrow>
<mml:mi>c</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mi>C</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> contains information about the oxygen concentration in the cage (in <inline-formula>
<mml:math display="inline" id="im11">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>m</mml:mi>
<mml:mi>g</mml:mi>
<mml:msup>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>). Effectively, each cage consists of a univariate time series. In the original dataset, other information was available, such as temperature or cage depth. However, in the experiments, variables other than oxygen concentration were not found to improve the hypoxia detection accuracy of the model. Therefore, these were excluded from the model.</p>
<p>The case study comprises data collected from 14 fish farms located in different places across eastern Canada. The data ranges over five years, from 2017 to 2023. The sampling period is detailed in <xref ref-type="table" rid="T1">
<bold>Table&#xa0;1</bold>
</xref>. The observations of each time series are captured with a high but, due to operator stocking plans, irregular sampling frequency. The period between consecutive observations can range from 1 to 3 minutes. We aim to forecast hourly values of oxygen concentration. Therefore, we aggregated the data in each cage to an hourly granularity. This is accomplished by taking the median of the collected values of each hour. If no observation was captured in a given hour this represented a missing observation. As an example, <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref> illustrates a sample of the time series of oxygen concentration in a given cage.</p>
<table-wrap id="T1" position="float">
<label>Table&#xa0;1</label>
<caption>
<p>General information about the fish farms in the case study.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Farm</th>
<th valign="top" align="left">Start</th>
<th valign="top" align="left">End</th>
<th valign="top" align="left"># Hourly observations</th>
<th valign="top" align="left"># Cages</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SI</td>
<td valign="top" align="left">2017&#x2013;10-19</td>
<td valign="top" align="left">2022&#x2013;09-19</td>
<td valign="top" align="right">402226</td>
<td valign="top" align="right">12</td>
</tr>
<tr>
<td valign="top" align="left">LP</td>
<td valign="top" align="left">2017&#x2013;11-10</td>
<td valign="top" align="left">2022&#x2013;09-09</td>
<td valign="top" align="right">521250</td>
<td valign="top" align="right">14</td>
</tr>
<tr>
<td valign="top" align="left">BIS</td>
<td valign="top" align="left">2018&#x2013;01-01</td>
<td valign="top" align="left">2023&#x2013;03-21</td>
<td valign="top" align="right">201387</td>
<td valign="top" align="right">13</td>
</tr>
<tr>
<td valign="top" align="left">MI</td>
<td valign="top" align="left">2019&#x2013;05-20</td>
<td valign="top" align="left">2022&#x2013;06-17</td>
<td valign="top" align="right">135900</td>
<td valign="top" align="right">13</td>
</tr>
<tr>
<td valign="top" align="left">DL</td>
<td valign="top" align="left">2019&#x2013;07-04</td>
<td valign="top" align="left">2022&#x2013;12-02</td>
<td valign="top" align="right">111793</td>
<td valign="top" align="right">7</td>
</tr>
<tr>
<td valign="top" align="left">RI</td>
<td valign="top" align="left">2020&#x2013;08-19</td>
<td valign="top" align="left">2022&#x2013;12-02</td>
<td valign="top" align="right">97481</td>
<td valign="top" align="right">9</td>
</tr>
<tr>
<td valign="top" align="left">VB</td>
<td valign="top" align="left">2021&#x2013;01-14</td>
<td valign="top" align="left">2023&#x2013;03-21</td>
<td valign="top" align="right">48200</td>
<td valign="top" align="right">6</td>
</tr>
<tr>
<td valign="top" align="left">CI</td>
<td valign="top" align="left">2021&#x2013;03-01</td>
<td valign="top" align="left">2023&#x2013;03-21</td>
<td valign="top" align="right">68397</td>
<td valign="top" align="right">5</td>
</tr>
<tr>
<td valign="top" align="left">AD</td>
<td valign="top" align="left">2021&#x2013;03-23</td>
<td valign="top" align="left">2022&#x2013;03-03</td>
<td valign="top" align="right">31810</td>
<td valign="top" align="right">6</td>
</tr>
<tr>
<td valign="top" align="left">SC</td>
<td valign="top" align="left">2021&#x2013;06-15</td>
<td valign="top" align="left">2022&#x2013;04-04</td>
<td valign="top" align="right">14792</td>
<td valign="top" align="right">5</td>
</tr>
<tr>
<td valign="top" align="left">DH</td>
<td valign="top" align="left">2021&#x2013;06-16</td>
<td valign="top" align="left">2022&#x2013;12-02</td>
<td valign="top" align="right">54105</td>
<td valign="top" align="right">6</td>
</tr>
<tr>
<td valign="top" align="left">RB</td>
<td valign="top" align="left">2021&#x2013;08-13</td>
<td valign="top" align="left">2023&#x2013;03-21</td>
<td valign="top" align="right">40790</td>
<td valign="top" align="right">6</td>
</tr>
<tr>
<td valign="top" align="left">FB</td>
<td valign="top" align="left">2021&#x2013;10-04</td>
<td valign="top" align="left">2022&#x2013;12-02</td>
<td valign="top" align="right">50729</td>
<td valign="top" align="right">6</td>
</tr>
<tr>
<td valign="top" align="left">CC</td>
<td valign="top" align="left">2022&#x2013;01-06</td>
<td valign="top" align="left">2022&#x2013;12-11</td>
<td valign="top" align="right">32640</td>
<td valign="top" align="right">4</td>
</tr>
</tbody>
</table>
</table-wrap>
<fig id="f1" position="float">
<label>Figure&#xa0;1</label>
<caption>
<p>Sample of the oxygen concentration time series in a cage within the LP fish farm. The horizontal dashed line denotes the threshold below which hypoxia occurs (7 <italic>mgL<sup>&#x2212;</sup>
</italic>
<sup>1</sup>).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g001.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f2">
<bold>Figure&#xa0;2</bold>
</xref> shows the number of data points over each month by fish farm, from October 2017 to December 2022. Until 2019, the data came from only two farms (SI and LP). The data for the year 2022 comprises the most data points, while there is a period with very little data during 2018.</p>
<fig id="f2" position="float">
<label>Figure&#xa0;2</label>
<caption>
<p>Number of data points by month and fish farm.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g002.tif"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Exploratory data analysis</title>
<p>We carried out an extensive exploratory data analysis to draw insights about the oxygen concentration dynamics. We illustrate the main findings in this section. <xref ref-type="fig" rid="f3">
<bold>Figure&#xa0;3</bold>
</xref> shows the oxygen concentration over time at each farm. For visualization purposes, we aggregated the data in each farm by taking the median across all available cages. Overall, this plot indicates the presence of yearly seasonality influencing oxygen concentration.</p>
<fig id="f3" position="float">
<label>Figure&#xa0;3</label>
<caption>
<p>Hourly oxygen concentration time series by farm. The values of the time series are aggregated (median) by cage. The plot suggests similar dynamics across farms and yearly seasonality.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g003.tif"/>
</fig>
<p>There are some periods of missing data, especially in the years 2018 and 2019. The mechanisms behind the missingness are either <italic>missing completely at random</italic> (e.g. random sensor malfunction in a given observation) or <italic>missing at random</italic> (e.g. sensor shutdown during some period for maintenance), according to the definitions described by <xref ref-type="bibr" rid="B21">Schafer and Graham (2002)</xref>.</p>
<p>We explored the distribution of oxygen concentration using violin plots in <xref ref-type="fig" rid="f4">
<bold>Figure&#xa0;4</bold>
</xref>. For this analysis, we studied each farm as a whole and disregarded information about individual cages. Overall, the distribution varies across farms. Besides the different locations of each farm, one possible reason for this variability was the sampling period at each farm is different, covering various lengths of time and capturing data during different seasons which is known to influence oxygen fluctuations (<xref ref-type="bibr" rid="B5">Burke et&#xa0;al., 2021</xref>).</p>
<fig id="f4" position="float">
<label>Figure&#xa0;4</label>
<caption>
<p>Distribution of oxygen concentration (<italic>mgL<sup>&#x2212;</sup>
</italic>
<sup>1</sup>) by farm.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g004.tif"/>
</fig>
<p>This work is devoted to the prediction of hypoxia events, which can be defined as an observation with an oxygen concentration below 7 <italic>mgL<sup>&#x2212;</sup>
</italic>
<sup>1</sup> (<xref ref-type="bibr" rid="B18">Oppedal et&#xa0;al., 2011</xref>). In this context, we analyzed the distribution of the oxygen condition (normal or hypoxia) across fish farms and over different months. <xref ref-type="fig" rid="f5">
<bold>Figure&#xa0;5</bold>
</xref> shows the relative frequency (%) of each condition across all observations at each fish farm. Hypoxia is generally a rare occurrence across all farms; however, events are more common at some farms such as DH and SI compared to others. <xref ref-type="fig" rid="f6">
<bold>Figure&#xa0;6</bold>
</xref> also shows that hypoxia is more common during the summer and fall seasons. This is expected as these events are correlated with higher water temperatures, which reduces the solubility of oxygen in seawater (<xref ref-type="bibr" rid="B5">Burke et&#xa0;al., 2021</xref>).</p>
<fig id="f5" position="float">
<label>Figure&#xa0;5</label>
<caption>
<p>Distribution of oxygen concentration condition across fish farms. Hypoxia (<italic>&lt;</italic> 7<italic>mgL<sup>&#x2212;</sup>
</italic>
<sup>1</sup>) conditions are more common in some farms than others.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g005.tif"/>
</fig>
<fig id="f6" position="float">
<label>Figure&#xa0;6</label>
<caption>
<p>Distribution of oxygen concentration condition across each month. Hypoxic (&lt; 7<italic>mgL</italic>
<sup>-1</sup>) conditions are more common in the summer and fall seasons.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g006.tif"/>
</fig>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Background and problem definition</title>
<p>This section describes the background of this work. We start by formalizing the forecasting task as an auto-regressive problem (Section 3.1). Then, we explain how this formalization extends to the exceedance probability estimation setting (Section 3.3). We also describe how global forecasting models operate and how they relate to our work (Section 3.2). Finally, in Section 3.4, we carry out a brief literature review and describe previous efforts in forecasting oxygen dynamics. We remark that the background provided here focuses on statistical and machine learning models, and does not consider deterministic biogeochemical models that also have been used to predict oxygen at salmon farm sites (e.g (<xref ref-type="bibr" rid="B29">Wild-Allen et&#xa0;al., 2020</xref>)).</p>
<sec id="s3_1">
<label>3.1</label>
<title>Time series forecasting</title>
<p>The general goal behind forecasting is to predict the value of the upcoming observations of oxygen concentration in a given cage, <inline-formula>
<mml:math display="inline" id="im12">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, given the historical data, where <italic>h</italic> denotes the forecasting horizon.</p>
<p>We formalize the problem of time series forecasting based on an auto-regressive strategy. Accordingly, observations of a time series are modeled based on their recent lags. More precisely, the value of <italic>c<sub>i</sub>
</italic> is modeled based on the most recent <inline-formula>
<mml:math display="inline" id="im13">
<mml:mi>q</mml:mi>
</mml:math>
</inline-formula> timesteps: <inline-formula>
<mml:math display="inline" id="im14">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mo>&#x2026;</mml:mo>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im15">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the observation we want to predict and <inline-formula>
<mml:math display="inline" id="im16">
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2208;</mml:mo>
<mml:mi mathvariant="script">X</mml:mi>
<mml:mo>&#x2282;</mml:mo>
<mml:msup>
<mml:mi>&#x211d;</mml:mi>
<mml:mi>p</mml:mi>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> represents the corresponding input lagged features (<xref ref-type="bibr" rid="B3">Bontempi et&#xa0;al., 2013</xref>). This approach leads to a multiple regression problem where the temporal dependency is modeled by using past observations as explanatory variables. Using <italic>h</italic> = 1 for illustration purposes, the goal is to transform each time series into a matrix structure as follows:</p>
<disp-formula>
<mml:math display="block" id="M1">
<mml:mrow>
<mml:mi>D</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mn>2</mml:mn>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>p</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mo>&#x22ef;</mml:mo>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>p</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>&#x22ee;</mml:mo>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:msub>
<mml:mi>t</mml:mi>
<mml:mi>C</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:math>
</disp-formula>
<p>Each row in the matrix above is a sample (X,c). The last column of the matrix denotes the target variable (future observations) and the remaining columns represent the lagged explanatory variables (<italic>X</italic>). <inline-formula>
<mml:math display="inline" id="im17">
<mml:mrow>
<mml:msub>
<mml:mi>D</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denotes the reconstructed time series in a matrix structure for a given cage <inline-formula>
<mml:math display="inline" id="im18">
<mml:mrow>
<mml:msub>
<mml:mi>C</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. For each forecasting horizon <inline-formula>
<mml:math display="inline" id="im19">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>, the goal is to build a multiple regression model <inline-formula>
<mml:math display="inline" id="im20">
<mml:mrow>
<mml:mi>f</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> that can be written as <inline-formula>
<mml:math display="inline" id="im21">
<mml:mrow>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mrow>
<mml:mi>i</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>h</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>X</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math display="inline" id="im22">
<mml:mrow>
<mml:msub>
<mml:mi>Z</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represents additional covariates known at the <italic>i</italic>-th instance. These covariates can be, for example, explanatory time series related to the phenomenon. For example, the season of the year.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Global forecasting models</title>
<p>Traditionally, forecasting models are trained using the historical data of the time series of interest, often referred to as local models (<xref ref-type="bibr" rid="B13">Januschowski et&#xa0;al., 2020</xref>). In our case, we need to forecast time series from several cages that belong to multiple fish farms. Leveraging the historical observations of all available time series can be valuable to build a forecasting model. The dynamics of the time series are often related, and a model may be able to learn useful patterns from some time series that have not revealed themselves in others. Models that are trained on multiple time series are often referred to as global forecasting models (<xref ref-type="bibr" rid="B11">Godahewa et&#xa0;al., 2021</xref>).</p>
<p>The popularity of global forecasting methods has surged following their successful use in the M4 forecasting competition, which featured 100000 time series from different application domains. The contest was won by a global forecasting method developed by <xref ref-type="bibr" rid="B22">Smyl (2020)</xref> that combines an exponential smoothing method with a recurrent neural network. Since then, several other works have been developed for training forecasting models with multiple time series (<xref ref-type="bibr" rid="B11">Godahewa et&#xa0;al., 2021</xref>).</p>
<p>The key motivation for our use of a global forecasting approach is giving the learning algorithm access to additional data. Machine learning algorithms tend to perform better with larger training sets, especially those with many parameters (<xref ref-type="bibr" rid="B7">Cerqueira et&#xa0;al., 2022</xref>). We will build a global model to forecast oxygen concentration based on the available information across all cages from all fish farms.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Exceedance probability forecasting</title>
<p>Conveying the uncertainty around forecasts is critical for better decision-making. Uncertainty in numerical forecasts is usually quantified by predicting intervals (<xref ref-type="bibr" rid="B14">Khosravi et&#xa0;al., 2011</xref>), or probabilities (<xref ref-type="bibr" rid="B10">Gneiting and Katzfuss, 2014</xref>).</p>
<p>For binary events, we can simply output the probability of that event occurring. An instance of a binary event is exceedance, which, as mentioned before, refers to when a time series exceeds a predefined threshold.</p>
<p>The probability of a hypoxia event <italic>h<sub>i</sub>
</italic> denotes the probability that the oxygen concentration <italic>c<sub>i</sub>
</italic> falls below 7 <italic>mgL<sup>&#x2212;</sup>
</italic>
<sup>1</sup> in any given instant <italic>i</italic>. In this type of problem, <italic>h<sub>i</sub>
</italic> is usually modeled by resorting to a binary target variable <italic>b<sub>i</sub>
</italic>, which can be formalized as shown in <xref ref-type="disp-formula" rid="eq1">
<bold>Equation 1</bold>
</xref>:</p>
<disp-formula id="eq1">
<label>(1)</label>
<mml:math display="block" id="M2">
<mml:mrow>
<mml:msub>
<mml:mi>b</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable columnalign="left">
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mn>1</mml:mn>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mtext>if&#x2009;</mml:mtext>
<mml:msub>
<mml:mi>c</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>&#x2264;</mml:mo>
<mml:mn>7</mml:mn>
<mml:mi>m</mml:mi>
<mml:mi>g</mml:mi>
<mml:msup>
<mml:mi>L</mml:mi>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mo>,</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr columnalign="left">
<mml:mtd columnalign="left">
<mml:mn>0</mml:mn>
</mml:mtd>
<mml:mtd columnalign="left">
<mml:mrow>
<mml:mtext>otherwise</mml:mtext>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>This setting leads to a classification problem where the goal is to build a classifier <italic>g</italic> of the form <italic>b<sub>i</sub>
</italic> = <italic>g</italic>(<italic>X<sub>i</sub>
</italic>). Logistic regression is commonly used for this purpose (<xref ref-type="bibr" rid="B25">Taylor and Yu, 2016</xref>). However, in our problem, a classification model would be unable to output the numeric oxygen forecasts which can be important for on-farm decision-making. Accordingly, in Section 4, we present an alternative approach to standard classification.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Related work</title>
<p>Understanding and forecasting water quality and oxygen dynamics is a relevant task and many works have been devoted to it (<xref ref-type="bibr" rid="B31">Zhang et&#xa0;al., 2022</xref>; <xref ref-type="bibr" rid="B12">Hai et&#xa0;al., 2023</xref>; <xref ref-type="bibr" rid="B32">Zhao et&#xa0;al., 2023</xref>). <xref ref-type="bibr" rid="B16">Liu et&#xa0;al. (2021)</xref> developed an ensemble based on Bayesian model averaging using a hybrid approach that includes artificial neural networks. Therein, the authors outline several state-of-the-art approaches to forecasting dissolved oxygen in the context of aquaculture.</p>
<p>Several works address the problem of detecting hypoxic events using data-driven methods. <xref ref-type="bibr" rid="B1">Arepalli and Naik (2024)</xref> tackle this problem using a deep learning method based on self-attention and LSTM layers. Besides dissolved oxygen, they also use other input variables such as temperature or turbidity. They report 99.8% accuracy, though this metric is usually a poor choice for problems involving imbalanced distributions (<xref ref-type="bibr" rid="B4">Bronco et&#xa0;al., 2016</xref>). Similar to us, <xref ref-type="bibr" rid="B19">Politicos et&#xa0;al. (2021)</xref> modeled hypoxia using tree-based models and reported an F1 score ranging between 0.84 and 0.93. The F1 score is a classification metric that combines precision and recall measures. However, while we deal with North Atlantic ocean data, the data used in their study was collected from an enclosed lagoon in the Greek Mediterranean.</p>
<p>We aim to predict hypoxic events and forecast oxygen concentration using a single model, whereas previous research addresses only one of these problems. As mentioned before, coupling forecasts with hypoxia probabilities can be useful to foster more informed decision-making by farmers.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Methodology</title>
<p>In this section, we describe the proposed approach for estimating the probability of hypoxia events in fish farms. We start by describing the forecasting approach based on a global model and how it is used to compute probabilistic estimates (Section 4.1). Then, we explain how the proposed method is evaluated (Section 4.2).</p>
<sec id="s4_1">
<label>4.1</label>
<title>Forecasting approach</title>
<p>The development of the model works in three main steps:</p>
<list list-type="order">
<list-item>
<p>Data preparation;</p>
</list-item>
<list-item>
<p>Building a forecasting model based on a global strategy;</p>
</list-item>
<list-item>
<p>Convert the forecasts produced by the model into probability estimates of hypoxia.</p>
</list-item>
</list>
<p>In the next subsections, we explain each step in turn.</p>
<sec id="s4_1_1">
<label>4.1.1</label>
<title>Data preparation</title>
<p>Several data preparation steps were carried out before training a model. The workflow is summarized in <xref ref-type="fig" rid="f7">
<bold>Figure&#xa0;7</bold>
</xref>. During the data preparation stage, we conducted a time series analysis using hypothesis testing to assess the presence of trend, seasonality, and heteroscedasticity in the data concerning each cage.</p>
<fig id="f7" position="float">
<label>Figure&#xa0;7</label>
<caption>
<p>Data preparation workflow.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g007.tif"/>
</fig>
<p>Regarding trend, we used the KPSS test (<xref ref-type="bibr" rid="B15">Kwiatkowski et&#xa0;al., 1992</xref>) to assess if differencing was required for stationarity. We conducted yearly seasonal differencing before applying the hypothesis test to control for seasonal effects. The column <italic>Trend</italic> in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> reports the ratio of cages in a given fish farm where the null hypothesis that the time series is trend-stationary is rejected with a significance value of 0.05. The null hypothesis is not rejected in all cases. However, in the interest of consistency and to keep all observations in a common value range, we took first differences in all-time series.</p>
<table-wrap id="T2" position="float">
<label>Table&#xa0;2</label>
<caption>
<p>Results of time series analysis for each fish farm.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="right">Farm</th>
<th valign="top" align="left">Trend</th>
<th valign="top" align="left">Seasonality</th>
<th valign="top" align="left">Variance</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="right">BIS</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">0.92</td>
</tr>
<tr>
<td valign="top" align="right">CI</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
</tr>
<tr>
<td valign="top" align="right">RB</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">0.83</td>
</tr>
<tr>
<td valign="top" align="right">VB</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
</tr>
<tr>
<td valign="top" align="right">DH</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
</tr>
<tr>
<td valign="top" align="right">DL</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
</tr>
<tr>
<td valign="top" align="right">FB</td>
<td valign="top" align="left">0.75</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">0.83</td>
</tr>
<tr>
<td valign="top" align="right">LP</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">0.79</td>
<td valign="top" align="left">1.0</td>
</tr>
<tr>
<td valign="top" align="right">MI</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">0.92</td>
</tr>
<tr>
<td valign="top" align="right">RI</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">1.0</td>
</tr>
<tr>
<td valign="top" align="right">SI</td>
<td valign="top" align="left">1.0</td>
<td valign="top" align="left">0.67</td>
<td valign="top" align="left">1.0</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<fn>
<p>The values represent the ratio of cages in the respective farm that require preprocessing on the corresponding component.</p>
</fn>
</table-wrap-foot>
</table-wrap>
<p>Visual inspection (c.f. <xref ref-type="fig" rid="f1">
<bold>Figure&#xa0;1</bold>
</xref>) indicated the presence of a strong yearly seasonal effect. We applied the heuristic by <xref ref-type="bibr" rid="B26">Wang et&#xa0;al. (2006)</xref> to measure the seasonal strength. They consider the presence of a relevant seasonal component if the seasonal strength is above 0.64<sup>1</sup>. The column <italic>Seasonality</italic> in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> reports the ratio of times that this occurs across each fish farm. The results corroborated the visual inspection. In effect, we attempted to account for seasonal variations using either Fourier series with yearly periods and monthly dummy variables. However, none of these approaches improved forecasting performance and were thus excluded from the model. We hypothesize that the first differences preprocessing operation has a sufficient stabilization effect and that it smooths out seasonal variations. Seasonality was checked in daily and weekly periodicity but the results did not give evidence for the presence of this component (results not showed).</p>
<p>Finally, we also checked whether each time series is heteroskedastic using the White test (<xref ref-type="bibr" rid="B28">White, 1980</xref>). The column <italic>Variance</italic> in <xref ref-type="table" rid="T2">
<bold>Table&#xa0;2</bold>
</xref> shows the ratio of cages in which the null hypothesis of equal variance was rejected with a significance value below 0.05. The results suggest that the time series are heteroskedastic, and we attempted to stabilize the variance of the data by taking the logarithm. This transformation did not improve forecasting accuracy, so it was also excluded from the final model.<xref ref-type="fn" rid="fn1">
<sup>1</sup>
</xref></p>
<p>After this analysis, we transformed each time series for supervised learning using the process outlined in Section 3.1. This involves computing lagged features for an auto-regressive modeling approach, with a forecasting horizon <italic>h</italic> from 1 to 24. The number of lags <italic>p</italic> is set to 24. This means that, at each time step, a model produces forecasts for each of the next 24 hours given the past 24 hours of data.</p>
<p>As we reported during the exploratory data analysis stage in Section 2.2, there are multiple observations with missing values in the cages across all fish farms. The missingness is due to either sensor malfunctions or sensor (or cage) maintenance periods and not related to the observed oxygen values. In effect, before the modeling stage we dropped the training samples that contain any missing value. More precisely, a training sample is the set of 24 lagged features plus the subsequent 24 observations to be predicted. If any of these 48 observations contain a missing value, the sample is discarded. In the testing stage, which is detailed in Section 4.2 below, we conduct a similar process. Essentially, we assume that the model works under the assumption that all lagged observations are available for computing predictions.</p>
</sec>
<sec id="s4_1_2">
<label>4.1.2</label>
<title>Modeling</title>
<p>We follow a global approach to build a forecasting model (c.f. Section 3.2). We built a single global model using the available data from all fish farms, as illustrated in <xref ref-type="fig" rid="f8">
<bold>Figure&#xa0;8</bold>
</xref>. The time series of each cage was preprocessed according to the process outlined in the previous section. Then, the available observations from all cages were concatenated into a single dataset that was used to train the model. We use the lightgbm method, a popular regression algorithm. This method was the backbone of the winning solution of the M5 forecasting competition <xref ref-type="bibr" rid="B17">Makridakis et&#xa0;al. (2022)</xref>, which also involved multiple time series (in that case, sales data of different retail products). The lightgbm is sensitive to different parameter configurations, so we optimized it using random search. The random search process was carried out with 200 iterations and using a validation set (explained below). The pool of parameters is detailed in <xref ref-type="table" rid="T3">
<bold>Table&#xa0;3</bold>
</xref>.</p>
<fig id="f8" position="float">
<label>Figure&#xa0;8</label>
<caption>
<p>Training and prediction workflow.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g008.tif"/>
</fig>
<table-wrap id="T3" position="float">
<label>Table&#xa0;3</label>
<caption>
<p>Pool of parameters for the lightgbm.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">ID</th>
<th valign="top" align="left">Parameter</th>
<th valign="top" align="left">Value</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">num_leaves</td>
<td valign="top" align="left">Max # of leaves per tree</td>
<td valign="top" align="left">{3, 5, 10, 15}</td>
</tr>
<tr>
<td valign="top" align="left">max_depth</td>
<td valign="top" align="left">Max depth per tree</td>
<td valign="top" align="left">{-1, 3, 5, 10, 15}</td>
</tr>
<tr>
<td valign="top" align="left">lambda_l1</td>
<td valign="top" align="left">L1 regularization</td>
<td valign="top" align="left">{0.1, 1, 10, 100}</td>
</tr>
<tr>
<td valign="top" align="left">lambda_l2</td>
<td valign="top" align="left">L2 regularization</td>
<td valign="top" align="left">{0.1, 1, 10, 100}</td>
</tr>
<tr>
<td valign="top" align="left">learning_rate</td>
<td valign="top" align="left">Learning rate</td>
<td valign="top" align="left">{0.05, 0.1, 0.2}</td>
</tr>
<tr>
<td valign="top" align="left">min_child_samples</td>
<td valign="top" align="left">Min # of points per leaf</td>
<td valign="top" align="left">{7, 15, 30}</td>
</tr>
<tr>
<td valign="top" align="left">boosting_type</td>
<td valign="top" align="left">Base algorithm</td>
<td valign="top" align="left">gbdt</td>
</tr>
<tr>
<td valign="top" align="left">num_boost_round</td>
<td valign="top" align="left">Boosting iterations</td>
<td valign="top" align="left">200</td>
</tr>
<tr>
<td valign="top" align="left">early_stopping_rounds</td>
<td valign="top" align="left">Early stopping</td>
<td valign="top" align="left">30</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We took a direct approach to multi-step forecasting. This means that we built a model for each forecasting horizon. This approach is a common alternative to the standard recursive method, which reuses a single model over the forecasting horizon. We prefer a direct method as it avoids the propagation of errors across the horizon (<xref ref-type="bibr" rid="B24">Taieb et&#xa0;al., 2012</xref>). After applying the model to get differenced predictions for a given time series, we revert the differencing operation to get the forecasts in the original scale.</p>
</sec>
<sec id="s4_1_3">
<label>4.1.3</label>
<title>Hypoxia probability estimates</title>
<p>Our goal is to forecast oxygen dynamics and predict the probability of an impending hypoxic event. Binary events such as hypoxia are usually modeled using a probabilistic binary classification model (c.f. Section 3.4). Instead, we adopt a strategy based on a forecasting model. We use the numeric forecasts for the upcoming instances to estimate the event probability according to the cumulative distribution function. In this cumulative distribution function, the location parameter of the distribution is derived from the output of the forecasting model, while the dispersion is estimated using the training data.</p>
<p>Let <inline-formula>
<mml:math display="inline" id="im23">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>c</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denote the forecast for a future <italic>i</italic>-th observation of oxygen concentration in a given cage produced by the lightgbm model. We assume that this prediction <inline-formula>
<mml:math display="inline" id="im24">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> can be modeled using a Normal distribution with mean <inline-formula>
<mml:math display="inline" id="im25">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and standard deviation <inline-formula>
<mml:math display="inline" id="im26">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>: <inline-formula>
<mml:math display="inline" id="im27">
<mml:mrow>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. The standard deviation <inline-formula>
<mml:math display="inline" id="im28">
<mml:mrow>
<mml:msub>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is computed using the historical observations of the corresponding cage.</p>
<p>In this context, we can estimate <inline-formula>
<mml:math display="inline" id="im29">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the hypoxia probability in a given cage at the <italic>i</italic>-th time step, using the cumulative distribution function (CDF) of <inline-formula>
<mml:math display="inline" id="im30">
<mml:mrow>
<mml:mo>&#xa0;</mml:mo>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> (<xref ref-type="disp-formula" rid="eq2">
<bold>Equation 2</bold>
</xref>):</p>
<disp-formula id="eq2">
<label>(2)</label>
<mml:math display="block" id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>h</mml:mi>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>C</mml:mi>
<mml:mi>D</mml:mi>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi mathvariant="script">N</mml:mi>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>y</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>i</mml:mi>
</mml:msub>
<mml:mo>,</mml:mo>
<mml:mtext>&#xa0;</mml:mtext>
<mml:msubsup>
<mml:mi>&#x3c3;</mml:mi>
<mml:mi>y</mml:mi>
<mml:mn>2</mml:mn>
</mml:msubsup>
</mml:mrow>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:msub>
<mml:mrow>
<mml:mo stretchy="false">(</mml:mo>
<mml:mi>&#x3c4;</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>When evaluated at the threshold <italic>&#x3c4;</italic>, the CDF represents the probability that the respective random variable will take a value less than or equal to <italic>&#x3c4;</italic>. In our case, the threshold <italic>&#x3c4;</italic> is equal to 7 <italic>mgL<sup>&#x2212;</sup>
</italic>
<sup>1</sup>.</p>
<p>The CDF-based approach to the problem is convenient because the same forecasting model can be used to predict both the upcoming values of oxygen concentration and to estimate hypoxia probability. While probabilistic forecasts are desirable for optimal decision-making, numeric forecasts may provide valuable insights as well that help prioritize resources. From an application standpoint, the classification approach simplifies the information being given to an operator, which may help a more inexperienced user with their decision-making, while a numeric forecast offers more information, which might be useful for a more experienced operator. For example, if it is near a threshold value, the predicted trajectory offers insights into how the variable is expected to change.</p>
<p>Our approach also enables threshold flexibility. A standard classification approach fixes the threshold <italic>&#x3c4;</italic> during training and it cannot be changed afterward. Our approach only uses the threshold during the prediction stage, so we can estimate the probability of an event at different thresholds. This can be important for farmers as the best threshold can depend on various factors, such as the number and age of fish in a cage. Overall, having access to the probabilities of an event at different thresholds might convey a better sense of the cost-benefits of available interventions.</p>
</sec>
</sec>
<sec id="s4_2">
<label>4.2</label>
<title>Evaluation</title>
<p>In this section, we detail the performance estimation procedure and evaluation metrics used.</p>
<sec id="s4_2_1">
<label>4.2.1</label>
<title>Estimation procedure</title>
<p>The data from different farms are collected in distinct periods, which makes it difficult to have a consistent testing period for all farms. In effect, we carried out a training and testing procedures for each fish farm where the testing period changes across these farms. For each fish farm, we set the last 40% of observations for testing. All previous observations for all farms (including the initial 60% of observations of the fish farm being tested) are compiled into a training set. The final testing periods for each fish farm are detailed in <xref ref-type="table" rid="T4">
<bold>Table&#xa0;4</bold>
</xref>.</p>
<table-wrap id="T4" position="float">
<label>Table&#xa0;4</label>
<caption>
<p>Training and testing periods across each farm.</p>
</caption>
<table frame="hsides">
<thead>
<tr>
<th valign="top" align="left">Farm</th>
<th valign="top" align="left">Testing Begin</th>
<th valign="top" align="left">Testing End</th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">SI</td>
<td valign="top" align="left">2020&#x2013;08-12 03:00:00</td>
<td valign="top" align="left">2022&#x2013;09-19 13:00:00</td>
</tr>
<tr>
<td valign="top" align="left">LP</td>
<td valign="top" align="left">2020&#x2013;10-31 03:00:00</td>
<td valign="top" align="left">2022&#x2013;09-09 10:00:00</td>
</tr>
<tr>
<td valign="top" align="left">BIS</td>
<td valign="top" align="left">2020&#x2013;12-02 10:00:00</td>
<td valign="top" align="left">2023&#x2013;03-21 12:00:00</td>
</tr>
<tr>
<td valign="top" align="left">MI</td>
<td valign="top" align="left">2021&#x2013;10-13 18:00:00</td>
<td valign="top" align="left">2022&#x2013;06-17 23:00:00</td>
</tr>
<tr>
<td valign="top" align="left">DL</td>
<td valign="top" align="left">2021&#x2013;11-25 06:00:00</td>
<td valign="top" align="left">2022&#x2013;12-02 21:00:00</td>
</tr>
<tr>
<td valign="top" align="left">RI</td>
<td valign="top" align="left">2022&#x2013;01-24 17:00:00</td>
<td valign="top" align="left">2022&#x2013;12-02 21:00:00</td>
</tr>
<tr>
<td valign="top" align="left">VB</td>
<td valign="top" align="left">2022&#x2013;07-12 02:00:00</td>
<td valign="top" align="left">2023&#x2013;03-21 12:00:00</td>
</tr>
<tr>
<td valign="top" align="left">CI</td>
<td valign="top" align="left">2022&#x2013;06-09 11:00:00</td>
<td valign="top" align="left">2023&#x2013;03-21 10:00:00</td>
</tr>
<tr>
<td valign="top" align="left">AD</td>
<td valign="top" align="left">2021&#x2013;10-10 00:00:00</td>
<td valign="top" align="left">2022&#x2013;03-03 12:00:00</td>
</tr>
<tr>
<td valign="top" align="left">SC</td>
<td valign="top" align="left">2021&#x2013;11-25 08:00:00</td>
<td valign="top" align="left">2022&#x2013;04-04 16:00:00</td>
</tr>
<tr>
<td valign="top" align="left">DH</td>
<td valign="top" align="left">2022&#x2013;05-20 17:00:00</td>
<td valign="top" align="left">2022&#x2013;12-02 20:00:00</td>
</tr>
<tr>
<td valign="top" align="left">RB</td>
<td valign="top" align="left">2022&#x2013;10-30 13:00:00</td>
<td valign="top" align="left">2023&#x2013;03-21 12:00:00</td>
</tr>
<tr>
<td valign="top" align="left">FB</td>
<td valign="top" align="left">2022&#x2013;06-15 07:00:00</td>
<td valign="top" align="left">2022&#x2013;12-02 21:00:00</td>
</tr>
<tr>
<td valign="top" align="left">CC</td>
<td valign="top" align="left">2022&#x2013;07-29 00:00:00</td>
<td valign="top" align="left">2022&#x2013;12-11 23:00:00</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>We also use a validation set for parameter optimization. This validation set is composed of the last 10% of observations of the training set. More precisely, the models are fit using the initial part of the training set (the initial 90% of observations of the complete training set) and evaluated in the validation set. After optimization, the model with the best parameters is then re-fit using the complete training set.</p>
</sec>
<sec id="s4_2_2">
<label>4.2.2</label>
<title>Evaluation metrics</title>
<p>In terms of metrics, we quantify the quality of the model from different perspectives. We use the absolute error to evaluate how well the forecasts approximate the actual oxygen concentration values. We evaluate the hypoxia probability estimates using log loss. This metric quantifies how close the estimated probability is from the actual value (0 if no event occurs, or 1 otherwise).</p>
<p>We also evaluate the predictions made by the model from an event detection point of view. Classification problems are usually evaluated using metrics such as precision, recall, and F1-score. However, these metrics can be misleading because, as <xref ref-type="bibr" rid="B9">Fawcett and Provost (1999)</xref> argue, they ignore the temporal order of observations and the value of timely predictions. The main goal of our work is to detect, in a timely manner, an impending hypoxic event. In effect, we resort to the evaluation approach proposed by <xref ref-type="bibr" rid="B27">Weiss and Hirsh (1998)</xref> and <xref ref-type="bibr" rid="B9">Fawcett and Provost (1999)</xref>. Specifically, we compute the event recall (ER) and the false alarm rate by unit of time (FAR). We can think of these metrics as variants of the standard recall and precision metrics but tailored for time-dependent data.</p>
<p>The ER metric can be defined as follows. Let <italic>T</italic> denote the total number of hypoxic events in a test dataset, and let <inline-formula>
<mml:math display="inline" id="im31">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>T</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> represent the total number of those events correctly predicted by a model <italic>m</italic>. The ER for model <italic>m</italic> is given by <xref ref-type="disp-formula" rid="eq3">
<bold>Equation 3</bold>
</xref>:</p>
<disp-formula id="eq3">
<label>(3)</label>
<mml:math display="block" id="M4">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>ER</mml:mtext>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>T</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:mfrac>
</mml:mrow>
</mml:math>
</disp-formula>
<p>ER differs from the classical recall metrics because a single correct prediction within an observation window leading to a hypoxic event is enough to consider that the event was correctly predicted.</p>
<p>The idea behind ER is that multiple alarms after the first one may add no value, assuming that action is taken after the initial alarm is issued. The same argument can be made for false alarms. The classical precision metric measures the ratio of positive predictions that are correct. Similarly to recall, in a time-dependent domain, the classical precision may be misleading because multiple predictions on the same event are counted multiple times. This idea is intuited in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>. This graphic shows a sequence in which predictions are being produced over time. Starting from time <italic>t<sub>i</sub>
</italic>, four false alarms are triggered. Performance evaluation should take the first wrong prediction into account as a false alarm. However, the subsequent false alarms (as shown in <xref ref-type="fig" rid="f9">
<bold>Figure&#xa0;9</bold>
</xref>) are not meaningful since they add no information.</p>
<fig id="f9" position="float">
<label>Figure&#xa0;9</label>
<caption>
<p>A sequence of consecutive false alarms. The first alarm is useful, but the subsequent ones may add no information.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g009.tif"/>
</fig>
<p>We compute the FAR, which can be defined as follows. First, we compute discounted false alarms (DFA) &#x2013; the number of non-overlapping observation periods associated with a wrong positive prediction. Then, we divide this number by the length of the time series. This leads to a measure of the expected number of false alarms per unit of time. In effect, the daily FAR for a model <italic>m</italic> is given by the <xref ref-type="disp-formula" rid="eq4">
<bold>Equation 4</bold>
</xref>:</p>
<disp-formula id="eq4">
<label>(4)</label>
<mml:math display="block" id="M5">
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>FAR</mml:mtext>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mfrac>
<mml:mrow>
<mml:msub>
<mml:mrow>
<mml:mtext>DFA</mml:mtext>
</mml:mrow>
<mml:mi>m</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mi>T</mml:mi>
</mml:mfrac>
<mml:mo>&#xd7;</mml:mo>
<mml:mn>24</mml:mn>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where <italic>T</italic> is the number of testing observations in the corresponding cage. We normalize the false alarm rate to a daily granularity in the interest of interpretability. This is accomplished by multiplying the original score (on an hourly basis) by 24.</p>
<p>Besides these metrics, we also analyze the average anticipation time of the model. That is, how long in advance the model is able to detect hypoxic events.</p>
</sec>
</sec>
</sec>
<sec id="s5">
<label>5</label>
<title>Experiments</title>
<p>In this section, we present the results of the experiments. We start by illustrating the output of the model (Section 5.1) and then evaluate it from different perspectives (Section 5.2).</p>
<sec id="s5_1">
<label>5.1</label>
<title>Visualizing the predictions</title>
<p>For illustration purposes, the model was tested on a sample cage in the LP fish farm and its predictions are shown in <xref ref-type="fig" rid="f10">
<bold>Figures&#xa0;10</bold>
</xref>, <xref ref-type="fig" rid="f11">
<bold>11</bold>
</xref>. <xref ref-type="fig" rid="f10">
<bold>Figure&#xa0;10</bold>
</xref> illustrates the forecasts for the next 24 hours, along with the actual and previous values of the series. In this particular example, the model can match the actual values closely.</p>
<fig id="f10" position="float">
<label>Figure&#xa0;10</label>
<caption>
<p>Sample of the forecasts for oxygen concentration, and respective true values, for a lead time of 12 hours.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g010.tif"/>
</fig>
<fig id="f11" position="float">
<label>Figure&#xa0;11</label>
<caption>
<p>Sample of the forecasts for low oxygen concentration events, and respective true values. The y-axis represents the probability of hypoxia computed with a lead time of 12 hours.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g011.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f11">
<bold>Figure&#xa0;11</bold>
</xref> shows the estimated probability of a hypoxic event in an example sample period. In it, a hypoxia event occurs in the final part of the sample (when the red line equals 1), and the model can detect it.</p>
</sec>
<sec id="s5_2">
<label>5.2</label>
<title>Forecasting accuracy across farms</title>
<p>This section presents a quantitative evaluation of the model described in Section 4.</p>
<p>We start by comparing the developed method that leverages forecasts into exceedance probability estimates using the CDF with a probabilistic binary classification model according to the log loss. The log loss evaluates how close the probability estimates are to the actual values (0 or 1). Having accurate probability estimates is important so that farmers can assess and react to a situation in a reliable way. The classification model is trained using a lightgbm using the same procedure we used for training the forecasting model (c.f. Section 4.1.2). The only change is that the classifier is optimized for binary classification rather than regression.</p>
<p>The results are shown in <xref ref-type="fig" rid="f12">
<bold>Figure&#xa0;12</bold>
</xref>. Overall, the developed model (denoted as CDF) shows better performance across almost all fish farms. We conducted a Wilcoxon signed rank test, recommended by <xref ref-type="bibr" rid="B8">Dem&#x161;ar (2006)</xref>, which confirmed that the difference in performance is significant.</p>
<fig id="f12" position="float">
<label>Figure&#xa0;12</label>
<caption>
<p>Log loss score across farms for the developed method (CDF) and a classification benchmark (Classifier).</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g012.tif"/>
</fig>
<p>
<xref ref-type="fig" rid="f13">
<bold>Figure&#xa0;13</bold>
</xref> shows the results for each farm in terms of event recall and false alarm rate, along with the number of hypoxia events that occurred during the testing period. The results vary considerably across fish farms. In some cases, the results are not representative due to the low number of events.</p>
<fig id="f13" position="float">
<label>Figure&#xa0;13</label>
<caption>
<p>Number of events, event recall, and false alarm rate by fish farm.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g013.tif"/>
</fig>
<p>(e.g., RB farm). For the farms with a higher number of events (e.g., LP, RI, and SI), the event recall ranges from 60% to 85%. The false alarm rate also varies across farms. For example, in the SI farm, the score is about 0.05. This means that we can expect around&#xa0;5 false alarm events (sequences of false alarms) every 100 days. High scores in terms of ER are associated with either a low number of events (which means the score is not representative) or a high FAR score. The end-users can control this trade-off using the decision threshold, which in these experiments is set to 0.5. Notwithstanding, this parameter can be optimized using cross-validation.</p>
<p>As mentioned before, the model is designed to make predictions with a lead time of 12 hours. However, sometimes the model might detect events with some delay. <xref ref-type="fig" rid="f14">
<bold>Figure&#xa0;14</bold>
</xref> shows the average anticipation time (in hours) for each fish farm. The results indicate that the model detects events with an anticipation time that ranges from about 5 to 8 hours.</p>
<fig id="f14" position="float">
<label>Figure&#xa0;14</label>
<caption>
<p>Average anticipation time (in hours) across all fish farms.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g014.tif"/>
</fig>
<p>Finally, we also evaluated the numeric forecasts produced by the model. <xref ref-type="fig" rid="f15">
<bold>Figure&#xa0;15</bold>
</xref> shows the distribution of absolute error (i.e., the absolute difference between the forecast and actual values) across the forecasting horizon. The median absolute error varies across farms. For example, this value is about 0.25 for the AD farm. Absolute errors above 0.5 are usually outliers (depending on the farm) according to the boxplot.</p>
<fig id="f15" position="float">
<label>Figure&#xa0;15</label>
<caption>
<p>Distribution of absolute error across farms.</p>
</caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="faquc-03-1365123-g015.tif"/>
</fig>
</sec>
</sec>
<sec id="s6" sec-type="discussion">
<label>6</label>
<title>Discussion</title>
<sec id="s6_1">
<label>6.1</label>
<title>On the data preparation and experiments</title>
<p>Pre-processing the data through differencing was found to be an important step in obtaining the reported forecasting performance, however, stabilizing the variance and modeling seasonal effects did not have a significant impact. We remark that, during the development of our model, we attempted to add extra meteorological and oceanographic variables. These included wind speed and direction, air temperature, solar radiation, or upwelling index. However, these did not improve forecasting performance. Another important aspect is that we average the values collected within each hour by taking the median. We adopted this approach to avoid potential anomalies due to sensor errors. However, we could also potentially have missed some brief hypoxia events that occur within each hour.</p>
<p>The learning algorithm applied in this work is a lightgbm that was optimized using random search. During the development stage, we attempted other algorithms, such as linear models and different types of neural networks. Overall, the lightgbm presented the best forecasting accuracy according to the metrics listed.</p>
<p>An important aspect of our work is that we evaluate our model based on realistic assumptions regarding its potential application. Usually, hypoxia detection models are evaluated using standard classification metrics such as precision, recall, and F1-score (e.g (<xref ref-type="bibr" rid="B1">Arepalli and Naik, 2024</xref>)). However, these metrics are not suitable for this problem due to the time-dependency among observations (c.f. Section 4.2.2).</p>
<p>The results suggest that the CDF-based approach for computing exceedance probability estimates outperforms a binary classification model. Beyond performance differences, it is important to note that the classification model is unable to produce forecasts, which are desirable in our application.</p>
<p>We could build two models, one regression model for forecasting and one classification model for exceedance probability estimation. However, two different models are susceptible to inconsistencies. For example, instances where the forecasts suggest a high chance of hypoxia but the classifier outputs a small probability for this event. In effect, being able to serve the two types of predictions using a single value provides consistency in the output. This can be important for a reliable use of the predictions by end-users.</p>
</sec>
<sec id="s6_2">
<label>6.2</label>
<title>On future work</title>
<p>There are several ways the model can be improved. We assume a constant variance when estimating the event probability using the CDF. Specifically, at each time step, we evaluate the CDF using the standard deviation of oxygen concentration of the corresponding cage computed with the whole training set. However, this statistic may vary over time, for example, due to seasonal effects. Thus, an adaptive approach (e.g., a rolling standard deviation) may lead to better probabilistic estimates.</p>
<p>Another point of improvement may be a better data compilation process. We apply a global approach by merging the training observations from all fish farms in a single dataset. This approach has several advantages such as the increased training sample size. This feature can be important if the user wants to deploy the model in fish farms with fewer data points available. However, introducing data from other farms may not bring value in some cases due to the idiosyncrasies of the farm. We can improve the global-local trade-off by, for example, clustering fish farms before modeling (<xref ref-type="bibr" rid="B2">Bandara et&#xa0;al., 2020</xref>).</p>
<p>Finally, the proposed model is based on auto-regression. We can also carry out an extended feature engineering process to improve the representation of the dataset. For example, previous works in the literature (e.g (<xref ref-type="bibr" rid="B16">Liu et&#xa0;al., 2021</xref>)) report that feature engineering based on empirical mode decomposition improves oxygen concentration forecasting accuracy.</p>
</sec>
</sec>
<sec id="s7" sec-type="conclusions">
<label>7</label>
<title>Conclusions</title>
<p>The key role of fish farms in efficiently producing marine protein and reducing pressure on wild fish populations underscores their importance in our society. The occurrence of hypoxia episodes requires a focused approach to mitigate the potential environmental and economic impacts of these events.</p>
<p>In this work, we have introduced a machine learning approach aimed at forecasting oxygen concentration and detecting hypoxic events in fish farms. Our solution is based on two steps. First, we build an auto-regressive global forecasting model, utilizing historical data from multiple fish farms. Subsequently, these forecasts are transformed into hypoxia probability estimates through the CDF. The empirical results, derived from the application of the proposed model across 14 diverse fish farms, show its effectiveness in detecting hypoxia.</p>
<p>We identified some potential limitations of our method. One area for improvement involves the representation of the dataset used for training the model. A more comprehensive feature engineering process could enhance the accuracy of the model. Additionally, we can also explore a better trade-off between employing a local a global forecasting model.</p>
</sec>
<sec id="s8" sec-type="data-availability">
<title>Data availability statement</title>
<p>The datasets presented in this article are not readily available because of privacy reasons. Requests to access the datasets should be directed to <uri xlink:href="https://jennie.korus@innovasea.com">jennie.korus@innovasea.com</uri>.</p>
</sec>
<sec id="s9" sec-type="author-contributions">
<title>Author contributions</title>
<p>VC: Conceptualization, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. JP: Data curation, Investigation, Software, Validation, Writing &#x2013; original draft, Writing &#x2013; review &amp; editing. JK: Data curation, Resources, Writing &#x2013; review &amp; editing. FB: Data curation, Resources, Software, Writing &#x2013; review &amp; editing. JA: Software, Validation, Writing &#x2013; review &amp; editing. MO: Methodology, Validation, Writing &#x2013; review &amp; editing. AS: Project administration, Resources, Supervision, Writing &#x2013; review &amp; editing. RF: Funding acquisition, Project administration, Resources, Supervision, Writing &#x2013; review &amp; editing. JG: Funding acquisition, Project administration, Resources, Supervision, Writing &#x2013; review &amp; editing. LT: Conceptualization, Funding acquisition, Project administration, Resources, Supervision, Writing &#x2013; original draft.</p>
</sec>
</body>
<back>
<sec id="s10" sec-type="funding-information">
<title>Funding</title>
<p>The author(s) declare financial support was received for the research, authorship, and/or publication of this article. This work was supported by the Atlantic Fisheries Fund project &#x201c;Big Data and Precision Fish Farming in Nova Scotia&#x201d; (AFF-NS-1544).</p>
</sec>
<ack>
<title>Acknowledgments</title>
<p>We thank Chris Whidden for taking the leadership on the project that supports this work, and Tyler Sclodnick for his valuable feedback.</p>
</ack>
<sec id="s11" sec-type="COI-statement">
<title>Conflict of interest</title>
<p>Author AS was employed by the company Cooke Aquaculture Inc.</p>
<p>Author JK was employed by the company Innovasea Marine Systems.</p>
<p>The remaining authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
<sec id="s12" sec-type="disclaimer">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<fn-group>
<fn id="fn1">
<label>1</label>
<p>see <ext-link ext-link-type="uri" xlink:href="https://www.rdocumentation.org/packages/forecast/versions/8.22.0/topics/nsdiffs">https://www.rdocumentation.org/packages/forecast/versions/8.22.0/topics/nsdiffs</ext-link>.</p>
</fn>
</fn-group>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Arepalli</surname> <given-names>P. G.</given-names>
</name>
<name>
<surname>Naik</surname> <given-names>K. J.</given-names>
</name>
</person-group> (<year>2024</year>). <article-title>A deep learning-enabled IoT framework for early hypoxia detection in aqua water using light weight spatially shared attention-LSTM network</article-title>. <source>J. Supercomputing</source>. <volume>80</volume>, <fpage>2718</fpage>&#x2013;<lpage>2747</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s11227-023-05580-x</pub-id>
</citation>
</ref>
<ref id="B2">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bandara</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Bergmeir</surname> <given-names>C.</given-names>
</name>
<name>
<surname>Smyl</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>Forecasting across time series databases using recurrent neural networks on groups of similar series: A clustering approach</article-title>. <source>Expert Syst. Appl.</source> <volume>140</volume>, <fpage>112896</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2019.112896</pub-id>
</citation>
</ref>
<ref id="B3">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Bontempi</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Ben Taieb</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Le Borgne</surname> <given-names>Y.-A.</given-names>
</name>
</person-group> (<year>2013</year>). &#x201c;<article-title>Machine learning strategies for time series forecasting</article-title>,&#x201d; in <source>Business Intelligence: Second European Summer School, eBISS 2012, Brussels, Belgium, July 15-21, 2012, Tutorial Lectures 2</source>, <fpage>62</fpage>&#x2013;<lpage>77</lpage>.</citation>
</ref>
<ref id="B4">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Bronco</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Torgo</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Ribeiro</surname> <given-names>R. P.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>A survey of predictive modeling on imbalanced domains</article-title>. <source>ACM Computing Surveys (CSUR)</source> <volume>49</volume>, <fpage>1</fpage>&#x2013;<lpage>50</lpage>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Burke</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Grant</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Filgueira</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Stone</surname> <given-names>T.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Oceanographic processes control dissolved oxygen variability at a commercial atlantic salmon farm: application of a real-time sensor network</article-title>. <source>Aquaculture</source> <volume>533</volume>, <fpage>736143</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aquaculture.2020.736143</pub-id>
</citation>
</ref>
<ref id="B6">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cerqueira</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Torgo</surname> <given-names>L.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>Exceedance probability forecasting via regression for significant wave height forecasting</article-title>. <source>arXiv</source> [preprint].</citation>
</ref>
<ref id="B7">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Cerqueira</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Torgo</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Soares</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A case study comparing machine learning with statistical methods for time series forecasting: size matters</article-title>. <source>J. Intelligent Inf. Syst.</source> <volume>59</volume>, <fpage>415</fpage>&#x2013;<lpage>433</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10844-022-00713-9</pub-id>
</citation>
</ref>
<ref id="B8">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Dem&#x161;ar</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Statistical comparisons of classifiers over multiple data sets</article-title>. <source>J. Mach. Learn. Res.</source> <volume>7</volume>, <fpage>1</fpage>&#x2013;<lpage>30</lpage>.</citation>
</ref>
<ref id="B9">
<citation citation-type="confproc">
<person-group person-group-type="author">
<name>
<surname>Fawcett</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Provost</surname> <given-names>F.</given-names>
</name>
</person-group> (<year>1999</year>). &#x201c;<article-title>Activity monitoring: Noticing interesting changes in behavior</article-title>,&#x201d; in <conf-name>Proceedings of the fifth ACM SIGKDD international conference on Knowledge discovery and data mining (ACM)</conf-name>. <fpage>53</fpage>&#x2013;<lpage>62</lpage>.</citation>
</ref>
<ref id="B10">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Gneiting</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Katzfuss</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2014</year>). <article-title>Probabilistic forecasting</article-title>. <source>Annu. Rev. Stat Its Appl.</source> <volume>1</volume>, <fpage>125</fpage>&#x2013;<lpage>151</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1146/annurev-statistics-062713-085831</pub-id>
</citation>
</ref>
<ref id="B11">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Godahewa</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Bandara</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Webb</surname> <given-names>G. I.</given-names>
</name>
<name>
<surname>Smyl</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Bergmeir</surname> <given-names>C.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Ensembles of localised models for time series forecasting</article-title>. <source>Knowledge-Based Syst.</source> <volume>233</volume>, <fpage>107518</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.knosys.2021.107518</pub-id>
</citation>
</ref>
<ref id="B12">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Hai</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>W.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Hu</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2023</year>). <article-title>A novel deep learning model for mining nonlinear dynamics in lake surface water temperature prediction</article-title>. <source>Remote Sens.</source> <volume>15</volume>, <fpage>900</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.3390/rs15040900</pub-id>
</citation>
</ref>
<ref id="B13">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Januschowski</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Gasthaus</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.</given-names>
</name>
<name>
<surname>Salinas</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Flunkert</surname> <given-names>V.</given-names>
</name>
<name>
<surname>Bohlke-Schneider</surname> <given-names>M.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <article-title>Criteria for classifying forecasting methods</article-title>. <source>Int. J. Forecasting</source> <volume>36</volume>, <fpage>167</fpage>&#x2013;<lpage>177</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ijforecast.2019.05.008</pub-id>
</citation>
</ref>
<ref id="B14">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Khosravi</surname> <given-names>A.</given-names>
</name>
<name>
<surname>Nahavandi</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Creighton</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Atiya</surname> <given-names>A. F.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Comprehensive review of neural network-based prediction intervals and new advances</article-title>. <source>IEEE Trans. Neural Networks</source> <volume>22</volume>, <fpage>1341</fpage>&#x2013;<lpage>1356</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1109/TNN.2011.2162110</pub-id>
</citation>
</ref>
<ref id="B15">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Kwiatkowski</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Phillips</surname> <given-names>P. C.</given-names>
</name>
<name>
<surname>Schmidt</surname> <given-names>P.</given-names>
</name>
<name>
<surname>Shin</surname> <given-names>Y.</given-names>
</name>
</person-group> (<year>1992</year>). <article-title>Testing the null hypothesis of stationarity against the alternative of a unit root: How sure are we that economic time series have a unit root</article-title>? <source>J. econometrics</source> <volume>54</volume>, <fpage>159</fpage>&#x2013;<lpage>178</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/0304-4076(92)90104-Y</pub-id>
</citation>
</ref>
<ref id="B16">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Liu</surname> <given-names>H.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Duan</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>A hybrid neural network model for marine dissolved oxygen concentrations time-series forecasting based on multi-factor analysis and a multi-model ensemble</article-title>. <source>Engineering</source> <volume>7</volume>, <fpage>1751</fpage>&#x2013;<lpage>1765</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eng.2020.10.023</pub-id>
</citation>
</ref>
<ref id="B17">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Makridakis</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Spiliotis</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Assimakopoulos</surname> <given-names>V.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>M5 accuracy competition: Results, findings, and conclusions</article-title>. <source>Int. J. Forecasting</source> <volume>38</volume>, <fpage>1346</fpage>&#x2013;<lpage>1364</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ijforecast.2021.11.013</pub-id>
</citation>
</ref>
<ref id="B18">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Oppedal</surname> <given-names>F.</given-names>
</name>
<name>
<surname>Dempster</surname> <given-names>T.</given-names>
</name>
<name>
<surname>Stien</surname> <given-names>L. H.</given-names>
</name>
</person-group> (<year>2011</year>). <article-title>Environmental drivers of atlantic salmon behaviour in sea-cages: a review</article-title>. <source>Aquaculture</source> <volume>311</volume>, <fpage>1</fpage>&#x2013;<lpage>18</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.aquaculture.2010.11.020</pub-id>
</citation>
</ref>
<ref id="B19">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Politicos</surname> <given-names>D. V.</given-names>
</name>
<name>
<surname>Petasis</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Katselis</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2021</year>). <article-title>Interpretable machine learning to forecast hypoxia in a lagoon</article-title>. <source>Ecol. Inf.</source> <volume>66</volume>, <fpage>101480</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2021.101480</pub-id>
</citation>
</ref>
<ref id="B20">
<citation citation-type="thesis">
<person-group person-group-type="author">
<name>
<surname>Remen</surname> <given-names>M.</given-names>
</name>
</person-group> (<year>2012</year>). <source>The oxygen requirement of atlantic salmon (salmo salar l.) in the on-growing phase in sea cages</source>. (<publisher-loc>Norway</publisher-loc>: <publisher-name>Institute of Biology, University of Bergen</publisher-name>).</citation>
</ref>
<ref id="B21">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Schafer</surname> <given-names>J. L.</given-names>
</name>
<name>
<surname>Graham</surname> <given-names>J. W.</given-names>
</name>
</person-group> (<year>2002</year>). <article-title>Missing data: our view of the state of the art</article-title>. <source>psychol. Methods</source> <volume>7</volume>, <fpage>147</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1037/1082-989X.7.2.147</pub-id>
</citation>
</ref>
<ref id="B22">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Smyl</surname> <given-names>S.</given-names>
</name>
</person-group> (<year>2020</year>). <article-title>A hybrid method of exponential smoothing and recurrent neural networks for time series forecasting</article-title>. <source>Int. J. Forecasting</source> <volume>36</volume>, <fpage>75</fpage>&#x2013;<lpage>85</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ijforecast.2019.03.017</pub-id>
</citation>
</ref>
<ref id="B23">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Subasinghe</surname> <given-names>R.</given-names>
</name>
<name>
<surname>Soto</surname> <given-names>D.</given-names>
</name>
<name>
<surname>Jia</surname> <given-names>J.</given-names>
</name>
</person-group> (<year>2009</year>). <article-title>Global aquaculture and its role in sustainable development</article-title>. <source>Rev. aquaculture</source> <volume>1</volume>, <fpage>2</fpage>&#x2013;<lpage>9</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/j.1753-5131.2008.01002.x</pub-id>
</citation>
</ref>
<ref id="B24">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taieb</surname> <given-names>S. B.</given-names>
</name>
<name>
<surname>Bontempi</surname> <given-names>G.</given-names>
</name>
<name>
<surname>Atiya</surname> <given-names>A. F.</given-names>
</name>
<name>
<surname>Sorjamaa</surname> <given-names>A.</given-names>
</name>
</person-group> (<year>2012</year>). <article-title>A review and comparison of strategies for multi-step ahead time series forecasting based on the nn5 forecasting competition</article-title>. <source>Expert Syst. Appl.</source> <volume>39</volume>, <fpage>7067</fpage>&#x2013;<lpage>7083</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.eswa.2012.01.039</pub-id>
</citation>
</ref>
<ref id="B25">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Taylor</surname> <given-names>J. W.</given-names>
</name>
<name>
<surname>Yu</surname> <given-names>K.</given-names>
</name>
</person-group> (<year>2016</year>). <article-title>Using auto-regressive logit models to forecast the exceedance probability for financial risk management</article-title>. <source>J. R. Stat. Society: Ser. A (Statistics Society)</source> <volume>179</volume>, <fpage>1069</fpage>&#x2013;<lpage>1092</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1111/rssa.12176</pub-id>
</citation>
</ref>
<ref id="B26">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Wang</surname> <given-names>X.</given-names>
</name>
<name>
<surname>Smith</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Hyndman</surname> <given-names>R.</given-names>
</name>
</person-group> (<year>2006</year>). <article-title>Characteristic-based clustering for time series data</article-title>. <source>Data Min. knowledge Discovery</source> <volume>13</volume>, <fpage>335</fpage>&#x2013;<lpage>364</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1007/s10618-005-0039-x</pub-id>
</citation>
</ref>
<ref id="B27">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Weiss</surname> <given-names>G. M.</given-names>
</name>
<name>
<surname>Hirsh</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>1998</year>). &#x201c;<article-title>Learning to predict rare events in event sequences</article-title>,&#x201d; In <source>KDD.</source> <volume>98</volume>, <fpage>359</fpage>&#x2013;<lpage>363</lpage>.</citation>
</ref>
<ref id="B28">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>White</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>1980</year>). <article-title>A heteroscedasticity-consistent covariance matrix estimator and a direct test for heteroscedasticity</article-title>. <source>Econometrica: J. Econometric Soc.</source> <fpage>817</fpage>&#x2013;<lpage>838</lpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.2307/1912934</pub-id>
</citation>
</ref>
<ref id="B29">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Wild-Allen</surname> <given-names>K.</given-names>
</name>
<name>
<surname>Andrewartha</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Baird</surname> <given-names>M.</given-names>
</name>
<name>
<surname>Bodrossy</surname> <given-names>L.</given-names>
</name>
<name>
<surname>Brewer</surname> <given-names>E.</given-names>
</name>
<name>
<surname>Eriksen</surname> <given-names>R.</given-names>
</name>
<etal/>
</person-group>. (<year>2020</year>). <source>Macquarie Harbour Oxygen Process model (FRDC 2016-067): CSIRO Final Report.</source> (<publisher-loc>Hobart, Australia</publisher-loc>: <publisher-name>CSIRO Oceans &amp; Atmosphere</publisher-name>). Available at: <uri xlink:href="https://www.frdc.com.au/sites/default/files/products/FRDC_MH_Final_Rep_June_2020.pdf">https://www.frdc.com.au/sites/default/files/products/FRDC_MH_Final_Rep_June_2020.pdf</uri>.</citation>
</ref>
<ref id="B30">
<citation citation-type="book">
<person-group person-group-type="author">
<name>
<surname>Yokoyama</surname> <given-names>H.</given-names>
</name>
</person-group> (<year>1997</year>). &#x201c;<article-title>Effects of fish farming on macroinvertebrates. comparison of three localities suffering from hypoxia</article-title>,&#x201d; in <source>Water Effluent and Quality, with Special Emphasis on Finfish and Shrimp Aquaculture</source> (<publisher-name>Texas A &amp; M University Sea Grant College Program</publisher-name>, <publisher-loc>Bryan, TX</publisher-loc>), <fpage>17</fpage>&#x2013;<lpage>24</lpage>.</citation>
</ref>
<ref id="B31">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Wang</surname> <given-names>Y.-G.</given-names>
</name>
<name>
<surname>Jeng</surname> <given-names>D.-S.</given-names>
</name>
<name>
<surname>Li</surname> <given-names>G.</given-names>
</name>
</person-group> (<year>2022</year>). <article-title>A physics-informed statistical learning framework for forecasting local suspended sediment concentrations in marine environment</article-title>. <source>Water Res.</source> <volume>218</volume>, <fpage>118518</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.watres.2022.118518</pub-id>
</citation>
</ref>
<ref id="B32">
<citation citation-type="journal">
<person-group person-group-type="author">
<name>
<surname>Zhao</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Yang</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Zhang</surname> <given-names>S.</given-names>
</name>
<name>
<surname>Wu</surname> <given-names>J.</given-names>
</name>
<name>
<surname>Zhao</surname> <given-names>Z.</given-names>
</name>
<name>
<surname>Jeng</surname> <given-names>D.-S.</given-names>
</name>
<etal/>
</person-group>. (<year>2023</year>). <article-title>Predictions of runoff and sediment discharge at the lower yellow river delta using basin irrigation data</article-title>. <source>Ecol. Inf.</source> <volume>78</volume>, <fpage>102385</fpage>. doi:&#xa0;<pub-id pub-id-type="doi">10.1016/j.ecoinf.2023.102385</pub-id>
</citation>
</ref>
</ref-list>
</back>
</article>