<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="brief-report">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Mar. Sci.</journal-id>
<journal-title>Frontiers in Marine Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Mar. Sci.</abbrev-journal-title>
<issn pub-type="epub">2296-7745</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fmars.2017.00082</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Marine Science</subject>
<subj-group>
<subject>Perspective</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>A Tale of Two Crowds: Public Engagement in Plankton Classification</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Robinson</surname> <given-names>Kelly L.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x0002A;</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/386241/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Luo</surname> <given-names>Jessica Y.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="author-notes" rid="fn004"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/410404/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Sponaugle</surname> <given-names>Su</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/423980/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Guigand</surname> <given-names>Cedric</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Cowen</surname> <given-names>Robert K.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Hatfield Marine Science Center, Oregon State University</institution> <country>Newport, OR, USA</country></aff>
<aff id="aff2"><sup>2</sup><institution>Department of Integrative Biology, Oregon State University</institution> <country>Corvallis, OR, USA</country></aff>
<aff id="aff3"><sup>3</sup><institution>Department of Marine Biology and Ecology, Rosenstiel School of Marine and Atmospheric Sciences, University of Miami</institution> <country>Miami, FL, USA</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Peng Liu, Institute of Remote Sensing and Digital Earth (CAS), China</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Guillem Chust, AZTI-Tecnalia, Spain; Yolanda F Wiersma, Memorial University of Newfoundland, Canada</p></fn>
<fn fn-type="corresp" id="fn001"><p>&#x0002A;Correspondence: Kelly L. Robinson <email>kelly.robinson&#x00040;louisiana.edu</email></p></fn>
<fn fn-type="other" id="fn002"><p>This article was submitted to Environmental Informatics, a section of the journal Frontiers in Marine Science</p></fn>
<fn fn-type="present-address" id="fn003"><p>&#x02020;Present Address: Kelly L. Robinson, Department of Biology, University of Louisiana at Lafayette, Lafayette, LA, USA;</p></fn>
<fn fn-type="present-address" id="fn004"><p>Jessica Y. Luo, Climate and Global Dynamics Lab, National Center for Atmospheric Research, Boulder, CO, USA</p></fn>
</author-notes>
<pub-date pub-type="epub">
<day>11</day>
<month>04</month>
<year>2017</year>
</pub-date>
<pub-date pub-type="collection">
<year>2017</year>
</pub-date>
<volume>4</volume>
<elocation-id>82</elocation-id>
<history>
<date date-type="received">
<day>19</day>
<month>10</month>
<year>2016</year>
</date>
<date date-type="accepted">
<day>10</day>
<month>03</month>
<year>2017</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2017 Robinson, Luo, Sponaugle, Guigand and Cowen.</copyright-statement>
<copyright-year>2017</copyright-year>
<copyright-holder>Robinson, Luo, Sponaugle, Guigand and Cowen</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) or licensor are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract><p>&#x0201C;Big data&#x0201D; are becoming common in biological oceanography with the advent of sampling technologies that can generate multiple, high-frequency data streams. Given the need for &#x0201C;big&#x0201D; data in ocean health assessments and ecosystem management, identifying and implementing robust, and efficient processing approaches is a challenge for marine scientists. Using a large plankton imagery data set, we present two crowd-sourcing approaches applied to the problem of classifying millions of organisms. The first used traditional crowd-sourcing by asking the public to identify plankton through a web-interface. The second challenged the data science community to develop algorithms via an industry partnership. We found traditional crowd-sourcing was an excellent way to engage and educate the public while crowd-sourcing data scientists rapidly generated multiple, effective solutions. As the need to process and visualize large and complex marine data sets is expected to grow over time, effective collaborations between oceanographers and computer and data scientists will become increasingly important.</p></abstract>
<kwd-group>
<kwd>marine science</kwd>
<kwd>plankton</kwd>
<kwd>big data</kwd>
<kwd>crowd-sourcing</kwd>
<kwd>machine learning</kwd>
<kwd>citizen science</kwd>
</kwd-group>
<contract-num rid="cn001">NSF-OCE 1419987</contract-num>
<contract-sponsor id="cn001">National Science Foundation<named-content content-type="fundref-id">10.13039/100000001</named-content></contract-sponsor>
<counts>
<fig-count count="2"/>
<table-count count="0"/>
<equation-count count="0"/>
<ref-count count="37"/>
<page-count count="7"/>
<word-count count="5157"/>
</counts>
</article-meta>
</front>
<body>
<sec id="s1">
<title>Data-intensive oceanography</title>
<p>Biological oceanography is rapidly becoming a data-intensive science with the advent of high resolution sampling technologies (Abbott, <xref ref-type="bibr" rid="B1">2013</xref>). Often referred to as the &#x0201C;fourth paradigm,&#x0201D; data-intensive science is the synergistic outcome of empirical, theoretical, and computational efforts that collect and analyze massive amounts of information from an array of sources (Gray, <xref ref-type="bibr" rid="B12">2009</xref>). It is made possible by the convergent evolution of high-resolution sensors, computing power, and networking capabilities which has accelerated the rate at which information about the environment can be gathered (Delaney and Barga, <xref ref-type="bibr" rid="B9">2009</xref>; Benson et al., <xref ref-type="bibr" rid="B3">2010</xref>; Porter et al., <xref ref-type="bibr" rid="B29">2012</xref>). Data emerging from this convergence are colloquially known as &#x0201C;big data&#x0201D;&#x02014;characterized by large volume, great variety, high veracity, and high velocity.</p>
<p>A key indicator of this shift to &#x0201C;big data&#x0201D; in biological oceanography is the precipitous rise in dataset size and complexity as a result of increased spatial, temporal, and taxonomic resolution, and increased rates of data generation (Table <xref ref-type="supplementary-material" rid="SM1">S1</xref>; Figure <xref ref-type="supplementary-material" rid="SM1">S1</xref>). Historically, large and complex biological oceanography datasets were generated exclusively by national- or international-level, multi-investigator programs that spanned years or even decades. The Challenger Expedition (the original &#x0201C;big data&#x0201D; program) collected 15,000 specimens representing 10,000 species at an average rate of 4,708 specimens year<sup>&#x02212;1</sup> over its 4-year (1872&#x02013;1876) circumnavigation of the world&#x00027;s oceans (Thomson and Murray, <xref ref-type="bibr" rid="B33">1895</xref>). The North Atlantic Continuous Plankton Recorder survey has collected 200 biological samples every month for the past 56 years, yielding nearly 2 million plankton records in total (Vezzulli and Reid, <xref ref-type="bibr" rid="B34">2003</xref>). Remote sensing programs like SeaWiFS (and its descendants Aqua MODIS, VIIRS, and MERIS) have gathered multiple types of global ocean color data daily at spatial resolutions ranging 10&#x02013;1,000 km for the past 20 years, resulting in a satellite imagery data set hundreds of terabytes in size (NASA Goddard Space Flight Center Ocean Biology Processing Group, <xref ref-type="bibr" rid="B25">2014</xref>). Recent efforts such as the Census of Marine Life (2000&#x02013;2010) have been able to individually assign 30.3 million marine organisms to one of 120,000 species at a rate of 3.3 million observations per year (Sedberry et al., <xref ref-type="bibr" rid="B30">2011</xref>), a nearly 900-fold increase when compared to the data generation rate of the Challenger Expedition for a similar set of questions.</p>
<p>The recent expansion and increased affordability of sampling technologies that can generate synoptic high-frequency observations for multiple data streams simultaneously has also allowed <italic>individual</italic> investigators to generate &#x0201C;big&#x0201D; data (Figure <xref ref-type="supplementary-material" rid="SM1">S1</xref>). These technologies include fixed and mobile remote sensing platforms, imaging systems, acoustic sensors, autonomous underwater vehicles (AUVs), and instruments used to generate genomics data. The flexibility to deploy many of these systems for long periods of time, either alone or as part of an integrated network, has allowed the measurement of processes and states at a range of spatial (10<sup>&#x02212;2</sup> to 10<sup>5</sup> m), temporal (millisecond to year), and taxonomic (species to domain) scales. For instance, plankton imaging systems such as the Video Plankton Recorder (VPR; Davis et al., <xref ref-type="bibr" rid="B7">1992</xref>), Underwater Vision Profiler (UVP; Picheral et al., <xref ref-type="bibr" rid="B28">2010</xref>), and now <italic>In Situ</italic> Ichthyoplankton Imaging System (ISIIS; Cowen and Guigand, <xref ref-type="bibr" rid="B4">2008</xref>) can resolve hundreds of thousands of individual organisms in a matter of hours, resulting in imagery datasets tens of terabytes in size. The ability to identify many of these organisms to low taxonomic levels adds layers of complexity to a dataset already spanning multiple temporal and spatial scales.</p>
<p>Thus, oceanographers and marine ecologists are increasingly finding themselves in a &#x0201C;deluge of data&#x0201D; (Baraniuk, <xref ref-type="bibr" rid="B2">2011</xref>), facing the challenge of robustly and efficiently storing, processing, and analyzing datasets that are voluminous, heterogeneous, and taxonomically complex (Figure <xref ref-type="supplementary-material" rid="SM1">S1</xref>). Previous discussions about working with &#x0201C;big&#x0201D; ecological data have focused largely on cyber-infrastructure capabilities, data management (Michener and Jones, <xref ref-type="bibr" rid="B23">2012</xref>; Gilbert et al., <xref ref-type="bibr" rid="B11">2014</xref>), and the need for data-driven approaches (Kelling et al., <xref ref-type="bibr" rid="B17">2009</xref>). While novel analytical techniques such as machine learning and crowd-sourcing for processing large and complex ecological data sets are increasingly reported in the terrestrial literature (Kelling et al., <xref ref-type="bibr" rid="B18">2013</xref>; Peters et al., <xref ref-type="bibr" rid="B27">2014</xref>), marine examples are limited (Wiley et al., <xref ref-type="bibr" rid="B36">2003</xref>; Dugan et al., <xref ref-type="bibr" rid="B10">2013</xref>; Millie et al., <xref ref-type="bibr" rid="B24">2013</xref>; Shamir et al., <xref ref-type="bibr" rid="B31">2014</xref>). Given this paucity and the need to use &#x0201C;big&#x0201D; biological oceanography and marine ecology data for rapid assessment of ocean health and adaptive management of ecosystems, we present here an evolution of approaches applied to the problem of efficiently classifying tens of millions of images of individual plankters generated by ISIIS. We discuss how partnerships were established, the questions we hoped to answer via crowd-sourcing approaches, as well as the success, the pit-falls, and the surprising outcomes generated by each approach.</p>
</sec>
<sec id="s2">
<title>Plankton imagery data</title>
<p>Plankton imagery data were collected with ISIIS, which is a towed underwater vehicle capable of imaging large volumes of water sufficient for quantifying rare mesoplankton (0.02&#x02013;20 mm) and macroplankton (2&#x02013;20 cm) such as larval fishes <italic>in situ</italic> (Cowen and Guigand, <xref ref-type="bibr" rid="B4">2008</xref>; Cowen et al., <xref ref-type="bibr" rid="B5">2013</xref>); Figure <xref ref-type="fig" rid="F1">1A</xref>). It is equipped with two line-scan cameras that can be deployed at either a fixed depth or set to &#x0201C;tow-yo&#x0201D; between two depths (see McClatchie et al., <xref ref-type="bibr" rid="B22">2012</xref>; Figure <xref ref-type="fig" rid="F1">1B</xref>). The large camera is capable of imaging 140 L s<sup>&#x02212;1</sup> with a 68 &#x003BC;m pixel resolution while the small camera can image 15 L s<sup>&#x02212;1</sup> at 59 &#x003BC;m pixel resolution. Each camera produces a continuous picture that then is parsed into equivalent frames at rates of 17 and 61 frames s<sup>&#x02212;1</sup>, respectively. Combined, these cameras produce 660 gigabytes of uncompressed imagery per hour. After each cruise, imagery data from each camera are pre-processed in a distributed computing environment. Each frame is flat-fielded to remove anomalous dark vertical bands and its contrast normalized to enhance regions with higher gray intensity. Regions-of-interest (i.e., particles) within each frame are then extracted from the frame using a process called segmentation that separates the signal (i.e., the particle) from the noise (i.e., frame background) (Figure <xref ref-type="fig" rid="F1">1C</xref>; Luo et al., in review). This process yields individual images, or &#x0201C;vignettes,&#x0201D; of planktonic organisms (Figure <xref ref-type="fig" rid="F1">1D</xref>). Because each frame typically contains multiple vignettes and ISIIS can be deployed for as many hours as there is sufficient storage, it can rapidly generate a dataset comprised of hundreds of thousands to millions of vignettes. The ISIIS has been successfully used in multiple ecosystems (McClatchie et al., <xref ref-type="bibr" rid="B22">2012</xref>; Greer et al., <xref ref-type="bibr" rid="B16">2013</xref>, <xref ref-type="bibr" rid="B15">2014</xref>, <xref ref-type="bibr" rid="B14">2015</xref>; Luo et al., <xref ref-type="bibr" rid="B21">2014</xref>). Data discussed below were collected in the North Atlantic, the Southern California Current, and the Straits of Florida.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p><bold>(A)</bold> Shipboard deployment of the <italic>In Situ</italic> Ichthyoplankton Imaging System (ISIIS). <bold>(B)</bold> Conceptual illustration of the ISIIS being towed during an undulation transect (i.e., &#x0201C;tow-yo&#x0201D;). <bold>(C)</bold> Example of a flat-fielded frame captured by the ISIIS. <bold>(D)</bold> Vignettes generated from the frame shown in <bold>(C)</bold> as a result of segmenting regions-of-interest.</p></caption>
<graphic xlink:href="fmars-04-00082-g0001.tif"/>
</fig>
</sec>
<sec id="s3">
<title>Manual classification</title>
<p>The first approach applied to ISIIS plankton imagery data was manual extraction and classification. For its first deployment in the Southern California Bight, a group of four experts manually extracted and identified to the lowest taxonomic level possible 85,000 organisms from 150,000 frames in 3 mo. (&#x02248;1440 person h; McClatchie et al. (<xref ref-type="bibr" rid="B22">2012</xref>). Greer et al. (<xref ref-type="bibr" rid="B16">2013</xref>) identified 35,028 gelatinous plankters and a subsample of highly abundant organisms such as copepods and appendicularians in 2 mo. Luo et al. (<xref ref-type="bibr" rid="B21">2014</xref>) expanded work by McClatchie et al. (<xref ref-type="bibr" rid="B22">2012</xref>) and hand-labeled 793,048 vignettes of gelatinous zooplankton found in 700,000 frames&#x02014;an effort that took an estimated 2,880 person h to complete over the course of 36 mo. It is noteworthy that these 3 years represent the time it took to identify only the gelatinous zooplankton and does not include abundant taxa such as copepods. Subsequent efforts to analyze ISIIS data collected over Stellwagen Bank and in the northern Gulf of Mexico utilized a combination of manual and semi-automatic classification methods (Greer et al., <xref ref-type="bibr" rid="B15">2014</xref>, <xref ref-type="bibr" rid="B13">2016</xref>). Greer et al. (<xref ref-type="bibr" rid="B15">2014</xref>) took &#x0007E;100 person h to hand-label 24,247 vignettes to either family (i.e., for larval fish) or class level (e.g., polychaetes, ctenophores, siphonophores). An automated particle counter was used to identify copepods; however, this latter process included manual sorting of an additional 12,981 vignettes for validation. Most recently, Greer et al. (<xref ref-type="bibr" rid="B13">2016</xref>) used a semi-automated classification method to sort 285,863 vignettes from 374,378 frames in 800 person h. For this effort, an identification expert manually extracted and classified each organism of interest with an ImageJ macro. Again, this approach did not include identifying the more abundant organisms such as copepods and appendicularians, as doing so would have greatly increased the amount of required classification time (Greer et al., <xref ref-type="bibr" rid="B13">2016</xref>). While these predominately manual classification methods yielded highly accurate results, the time required (2 mo to 1.5 years of classification effort per person) of a few experts to achieve this accuracy was considerable.</p>
</sec>
<sec id="s4">
<title>General crowd</title>
<p>Our next step was to use traditional crowd-sourcing (i.e., citizen science) to assist with ISIIS imagery plankton classification. Citizen science has long been an effective outreach and data collection tool for marine ecologists (e.g., coral reef monitoring, Pattengill-Semmens and Semmens, <xref ref-type="bibr" rid="B26">2003</xref>), assessing invasive species (Delaney et al., <xref ref-type="bibr" rid="B8">2008</xref>), tracking marine debris (Smith and Edgar, <xref ref-type="bibr" rid="B32">2014</xref>), and categorizing whale calls (Shamir et al., <xref ref-type="bibr" rid="B31">2014</xref>). For our project &#x0201C;Plankton Portal,&#x0201D; we partnered with Zooniverse (<ext-link ext-link-type="uri" xlink:href="http://www.zooniverse.org">www.zooniverse.org</ext-link>), one of the major hubs for online citizen science projects. Zooniverse is most well-known for its highly successful Galaxy Zoo project, where individuals helped astronomers identify planets, galaxies, and stars (Lintott et al., <xref ref-type="bibr" rid="B19">2008</xref>).</p>
<p>Development of the Plankton Portal (PP) site required extensive consideration of the simplest task users could be asked to do, and progressed in two phases (e.g., PP v. 1 and 2). For the original launch of PP v. 1, we decided that the target plankton taxa needed to be grouped into dominant shapes, as opposed to taxonomy. Therefore, our &#x0201C;round&#x0201D; category included two types of lobate ctenophores and pelagic tunicates. Our &#x0201C;elongated or ribbon-like&#x0201D; category included disparate organisms such as chaetognaths, radiolarian colonies, and cestid ctenophores. Likewise, our field guide (Available at: <ext-link ext-link-type="uri" xlink:href="https://github.com/Planktos/TwoCrowds">https://github.com/Planktos/TwoCrowds</ext-link>) reflected this thinking, where organisms were described according to their shape and appearance, as opposed to taxonomic identifiers. For PP v. 1, we asked users to: (1) identify the organism from a set of categories, (2) identify the organism&#x00027;s orientation, and (3) measure the organism&#x00027;s major and minor axes. However, with the second iteration (PP v. 2), which launched in June 2015 with the addition of a new dataset from the Mediterranean, we greatly simplified the classification tasks. Instead of measuring size and orientation as well as classification, users were asked to only classify objects. This simplification was made after observations that many users struggled with the orientation and measurement tasks. For example, users typically had trouble measuring the length of curved organisms (e.g., curved chaetognaths, larvaceans, and shrimp), as well as the orientation of organisms where specialized knowledge was needed to determine top vs. bottom (e.g., lobate and beroid ctenophores). Eliminating these tasks made classification faster and easier, as well as accessible on mobile devices.</p>
<p>Crowd-sourcing methods require the crowd, or group of users completing a classification task, to converge on a common answer. This convergence helped ensure data quality dimensions like precision and reliability are met (Wang and Strong, <xref ref-type="bibr" rid="B35">1996</xref>). To systematically implement such a rule, we established &#x0201C;retirement rules,&#x0201D; or a set of conditions that cause an image&#x00027;s classification to be deemed final. Our rules were as follows:</p>
<list list-type="order">
<list-item><p>If the first three people agree that the image is blank</p></list-item>
<list-item><p>If there are four classifications that state that the image is blank</p></list-item>
<list-item><p>If six people submit identical counts of taxa within an image</p></list-item>
<list-item><p>If the classification count reaches or exceeds 12</p></list-item>
</list>
<p>Despite the initial boost from Plankton Portal&#x00027;s launch (500,000 classifications in the first 6 mo.), classifications and unique users dropped off, though steady classification was still fueled by dedicated volunteers (Figure <xref ref-type="fig" rid="F2">2A</xref>). One such volunteer was quickly made a moderator, and as of September 2016, has contributed more than 275,000 classifications. Furthermore, the top 10% of users (3,146 out of 31,457) have contributed 83% of the classifications to date. However, as the rate of classifications dropped, the second 500,000 classifications took 1.5 years to complete. The combination of the exponential decline in classification rates with the necessary retirement rules resulted in only 24% of 403,881 frames completed (with 7% paused and 18% ongoing), despite receiving over 1.2 million classifications from over 10,000 users.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p><bold>(A)</bold> Cumulative plankton classifications made by citizen scientists engaged in Plankton Portal and cumulative number of unique users in Plankton Portal over time (year-month). <bold>(B)</bold> Precision of classifying algorithms for the top five teams participating in the inaugural National Data Science Bowl.</p></caption>
<graphic xlink:href="fmars-04-00082-g0002.tif"/>
</fig>
<p>The drop in classification rate and the rise in classifications by a relatively small user group prompted us to consider the outcomes of Plankton Portal relative to other, similar Zooniverse projects like Snapshot Serengeti. While many plankton groups like jellyfish may be charismatic, the average user typically is not familiar with their shapes, sizes, and diversity, resulting in a lower crowd information quality (Lukyanenko et al., <xref ref-type="bibr" rid="B20">2014</xref>). The presumed greater success of Snapshot Serengeti is likely attributed to its users possessing basic knowledge of African wildlife, as their characteristics are often taught during childhood. Thus, Snapshot Serengeti users become competent classifiers more quickly relative to Plankton Portal users. This reduced learning curve increases the rate of accurate classifications (and the user&#x00027;s confidence), and, ostensibly, results in continued participation, including engagement in more difficult tasks like resolving orientation. Thus, we propose that the efficacy of traditional citizen science projects will be greater when participants have an <italic>a priori</italic> an understanding or interest of the study components at some level, making specific and higher order tasks easier to learn and execute.</p>
</sec>
<sec id="s5">
<title>Specialized crowd</title>
<p>The most recent approach to the plankton image classification problem combined elements of traditional crowd-sourcing and algorithm development. Through our collaborators at Zooniverse, we were approached by Kaggle, an online data science competition community, and Booz Allen Hamilton, a management and technology consultant firm. These companies were looking for a compelling &#x0201C;big data&#x0201D; problem for their inaugural National Data Science Bowl (<ext-link ext-link-type="uri" xlink:href="http://www.datasciencebowl.com">www.datasciencebowl.com</ext-link>), a competition created to effect social good through data science analytics. Kaggle has hosted several competitions for marine scientists, including one in 2009 for cetacean biologists seeking an automated approach to identifying individual whale calls from bioacoustic data (Dugan et al., <xref ref-type="bibr" rid="B10">2013</xref>) as well as a recent competition to automate the recognition of right whales from images of rostrum callosity patterns (<ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/c/noaa-right-whale-recognition">https://www.kaggle.com/c/noaa-right-whale-recognition</ext-link>). For each of these competitions, the host provides an example data set that is used to train and test predictive algorithms created by competitors.</p>
<p>The National Data Science Bowl competition dataset consisted of 60,736 plankton vignettes separated into 121 classes (Cowen et al., <xref ref-type="bibr" rid="B6">2015</xref>). To create this dataset, approximately two million vignettes from the western Straits of Florida were manually sorted by five experts over 2 mo (&#x02248;650 person h). A subset of the two million was used because over 50% of the initial set were artifacts (e.g., bubbles, water movement wisps, and frame edges), and we wanted the number of vignettes in each class to represent a somewhat realistic frequency distribution of plankton types. For example, the number of copepod vignettes (7,732) far exceeded the number of larval fish vignettes (660). While this approach yielded a representative dataset, it meant rare, but highly important taxa such as larval fishes, would be under-represented during training. This under-representation can be problematic, since the efficacy at which a classification algorithm correctly identifies an image can vary with how often the program &#x0201C;sees&#x0201D; that class during training. To mediate, we targeted a minimum of 100 vignettes for taxonomic groups of particular interest.</p>
<p>Vignettes were at first classified based on strict taxonomy. This approach generated classes that consisted of organisms in all orientations, of varying sizes, and life stages. Vignettes were then partially re-sorted into classes that reflected broad taxonomic groups, organism shape (i.e., orientation), and size. Larval fish, for example, were separated into five classes: Very thin fish, thin fish, medium-bodied fish, and deep-bodied fish, myctophids, and leptocephali larvae.</p>
<p>The dataset was then split 30:70 into training and test sets. The training set was made available to competitors and was stripped of all metadata. Competitors were given 3 mo (December 15, 2014&#x02013;March 14, 2015) to develop the most accurate classification algorithm, measured by the lowest log-loss. Cash prizes (&#x00024;100,000&#x02013;1st, &#x00024;45,000&#x02013;2nd, &#x00024;15,000&#x02013;3rd, and &#x00024;15,000 to the top student team) provided the means to engage 1,293 participants comprising 1,049 teams from multiple countries. All submissions were made available on the competition website (<ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/c/datasciencebowl/submissions/all">https://www.kaggle.com/c/datasciencebowl/submissions/all</ext-link>) following the close of the competition. The top five algorithms had overall classification accuracies greater than 80%, a marked improvement in comparison to the Support Vector Machine (A. Sarafraz and C. Mader, University of Miami, pers. comm.) or Random Forest-based models previously used (ZooProcess; J.O. Irisson, University of Pierre and Marie Curie, pers. comm.). Mean precision (true positives/(true positives &#x0002B; false positives) was also high at 77%, although it varied among taxonomic groups (Figure <xref ref-type="fig" rid="F2">2B</xref>).</p>
</sec>
<sec id="s6">
<title>General vs. specialized crowd</title>
<p>Setting up our second crowd-sourcing approach as a competition to produce a product was advantageous over the traditional crowd-sourcing approach for multiple reasons. First, the competition yielded a suite of classifying algorithms we could use for future projects. This differs from traditional citizen science projects in that our crowd&#x00027;s productivity extended beyond the active engagement time. Second, we were able to quantitatively, comprehensively, and simultaneously evaluate a large suite of classification schemes in real-time. Third, it allowed us to benefit from group intelligence, a phenomenon where the collective intelligence of group is greater than the sum of the intelligences of individuals in the group (Woolley et al., <xref ref-type="bibr" rid="B37">2010</xref>). This benefit was realized two-fold since most participants were part of a team and many answered questions and shared ideas on the competition forum (<ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/c/datasciencebowl/forums">https://www.kaggle.com/c/datasciencebowl/forums</ext-link>). Fourth, the large cash prize meant participants were highly motivated. The limited timeline for submissions meant a rapid turn-around from problem to ultimate solution. Social advantages were also realized. The competition provided a platform for developing collaborative relationships between the top machine learning and computer vision researchers in the world and biological oceanographers. These collaborations have set off a cascade of activity at Oregon State University in the United States and at the University of Pierre and Marie Curie in France as solutions from the competition were quickly utilized to classify vignettes from a variety of plankton imaging systems.</p>
<p>Challenges associated with the competition were few. Most notable was that the structure of some of top solutions did not scale easily from the competition dataset (60,000 vignettes) to the actual size of our Straits of Florida imagery data set (&#x02248;340 million vignettes). This structural issue meant that additional programming expertise was required to significantly modify the algorithms before they could be applied. A second and related challenge was implementing the solutions, as many were created with cutting-edge techniques. Possible improvements to the competition framework could include a judging criteria regarding the scalability of the algorithm structure, as well as asking top competitors to consult for a period of time (e.g., 12 mo.) post-competition. Alternatively, research teams could include a bioinformatics specialist from the onset to facilitate the transfer and implementation of algorithms.</p>
</sec>
<sec sec-type="conclusions" id="s7">
<title>Conclusions</title>
<p>The archetypal approach of a single research group processing and analyzing large datasets in isolation is becoming increasingly infeasible&#x02014;particularly given the need for the data to be promptly incorporated into ocean health assessments and marine ecosystem management. An effective, alternative approach is citizen science. We found that traditional crowd-sourcing was an excellent way to engage and educate a broad spectrum of the public, while simultaneously applying human capital to a labor-intensive task. However, crowd-sourcing data scientists, a small and highly specialized group, was the most efficient approach to solving a complex bioinformatics problem. Our results highlight that traditional citizen science projects will be the most effective when the questions under study are already familiar to participants or the tasks the participants are asked to engage in are easy to learn. Lastly, as the need to process and visualize large and complex marine science data is expected to grow over time, collaborations between biological oceanographers, marine ecologists, and computer and data scientists will become increasingly valuable.</p>
</sec>
<sec id="s8">
<title>Author contributions</title>
<p>KR, JL, SS, and RC designed the National Data Science Bowl (NDSB) competition. JL and KR created the NDSB dataset. CG, JL, and RC designed and ran the Plankton Portal citizen science project. KR analyzed the &#x0201C;big data&#x0201D; trends data. JL contributed Plankton Portal data. KR, JL, SS, and RC wrote the manuscript.</p>
<sec>
<title>Conflict of interest statement</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
</sec>
</sec>
</body>
<back>
<ack>
<p>The authors wish to thank our National Data Science Bowl host Kaggle.com and sponsor Booz Allen Hamilton (in particular, Anthony Goldbloom, William Cukierski, Steven Mills, and Angela Zutavern). We also thank the Zooniverse team for the design and development of Plankton Portal, as well as Jean-Olivier Irisson, who provided data on the Plankton Portal&#x02014;Mediterranean section. We acknowledge the contributions of our volunteer moderators, particularly Zuzana Mach&#x000E1;&#x0010D;kov&#x000E1;. Lastly, we thank Clare Hansen, Megan Atkinson, and Mackenzie Mason for their assistance creating the NDSB competition image set. This manuscript was written with support from the National Science Foundation (NSF-OCE 1419987).</p>
</ack>
<sec sec-type="supplementary-material" id="s10">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="http://journal.frontiersin.org/article/10.3389/fmars.2017.00082/full#supplementary-material">http://journal.frontiersin.org/article/10.3389/fmars.2017.00082/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="DataSheet1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Abbott</surname> <given-names>M. R.</given-names></name></person-group> (<year>2013</year>). <article-title>From the President: the era of big data comes to oceanography</article-title>. <source>Oceanography</source> <volume>26</volume>, <fpage>7</fpage>&#x02013;<lpage>8</lpage>. <pub-id pub-id-type="doi">10.5670/oceanog.2013.68</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Baraniuk</surname> <given-names>R. G.</given-names></name></person-group> (<year>2011</year>). <article-title>More is less: signal processing and the data deluge</article-title>. <source>Science</source> <volume>331</volume>, <fpage>717</fpage>&#x02013;<lpage>719</lpage>. <pub-id pub-id-type="doi">10.1126/science.1197448</pub-id><pub-id pub-id-type="pmid">21311012</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Benson</surname> <given-names>B. J.</given-names></name> <name><surname>Bond</surname> <given-names>B. J.</given-names></name> <name><surname>Hamilton</surname> <given-names>M. P.</given-names></name> <name><surname>Monson</surname> <given-names>R. K.</given-names></name> <name><surname>Monson</surname> <given-names>R. K.</given-names></name> <name><surname>Han</surname> <given-names>R.</given-names></name></person-group> (<year>2010</year>). <article-title>Perspectives on next-generation technology for environmental sensor networks</article-title>. <source>Front. Ecol. Environ.</source> <volume>8</volume>, <fpage>193</fpage>&#x02013;<lpage>200</lpage>. <pub-id pub-id-type="doi">10.1890/080130</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cowen</surname> <given-names>R. K.</given-names></name> <name><surname>Guigand</surname> <given-names>C. M.</given-names></name></person-group> (<year>2008</year>). <article-title><italic>In situ</italic> Ichthyoplankton Imaging System (ISIIS): system design and preliminary results</article-title>. <source>Limnol. Oceanogr. Methods</source> <volume>6</volume>, <fpage>126</fpage>&#x02013;<lpage>132</lpage>. <pub-id pub-id-type="doi">10.4319/lom.2008.6.126</pub-id></citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cowen</surname> <given-names>R. K.</given-names></name> <name><surname>Greer</surname> <given-names>A. T.</given-names></name> <name><surname>Guigand</surname> <given-names>C. M.</given-names></name> <name><surname>Hare</surname> <given-names>J. A.</given-names></name> <name><surname>Richardson</surname> <given-names>D. E.</given-names></name> <name><surname>Walsh</surname> <given-names>H. J.</given-names></name></person-group> (<year>2013</year>). <article-title>Evaluation of the <italic>In Situ</italic> Ichthyoplankton Imaging System (ISIIS): comparison with the traditional (bongo net) sampler</article-title>. <source>Fish Bull</source>. <volume>111</volume>, <fpage>1</fpage>&#x02013;<lpage>12</lpage>. <pub-id pub-id-type="doi">10.7755/FB.111.1.1</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Cowen</surname> <given-names>R. K.</given-names></name> <name><surname>Sponaugle</surname> <given-names>S.</given-names></name> <name><surname>Robinson</surname> <given-names>K. L.</given-names></name> <name><surname>Luo</surname> <given-names>J. Y.</given-names></name></person-group> (<year>2015</year>). <source>Data from: PlanktonSet 1.0: Plankton Imagery Data Collected from F.G. Walton Smith in Straits of Florida from 2014-06-03 to 2014-06-06 and Used in the 2015 National Data Science Bowl.</source> NOAA National Centers for Environmental Information.</citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Davis</surname> <given-names>C. S.</given-names></name> <name><surname>Gallager</surname> <given-names>S. M.</given-names></name> <name><surname>Berman</surname> <given-names>M. S.</given-names></name> <name><surname>Haury</surname> <given-names>L. R.</given-names></name> <name><surname>Strickler</surname> <given-names>J. R.</given-names></name></person-group> (<year>1992</year>). <article-title>The Video Plankton Recorder (VPR): design and initial results</article-title>. <source>Arch. Hydrobiol. Beih</source> <volume>36</volume>, <fpage>67</fpage>&#x02013;<lpage>81</lpage>.</citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Delaney</surname> <given-names>D. G.</given-names></name> <name><surname>Sperling</surname> <given-names>C. D.</given-names></name> <name><surname>Adams</surname> <given-names>C. C.</given-names></name> <name><surname>Leung</surname> <given-names>B.</given-names></name></person-group> (<year>2008</year>). <article-title>Marine invasive species: validation of citizen science and implications for national monitoring networks</article-title>. <source>Biol. Invasions</source> <volume>10</volume>, <fpage>117</fpage>&#x02013;<lpage>128</lpage>. <pub-id pub-id-type="doi">10.1007/s10530-007-9114-0</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Delaney</surname> <given-names>J. R.</given-names></name> <name><surname>Barga</surname> <given-names>R. S.</given-names></name></person-group> (<year>2009</year>). <article-title>A 2020 vision for ocean science</article-title>, in <source>The Fourth Paradigm: Data-Intensive Scientific Discovery</source>, eds <person-group person-group-type="editor"><name><surname>Hey</surname> <given-names>T.</given-names></name> <name><surname>Tansley</surname> <given-names>S.</given-names></name> <name><surname>Tolle</surname> <given-names>K.</given-names></name></person-group> (<publisher-loc>Redmond, WA</publisher-loc>: <publisher-name>Microsoft Research</publisher-name>), <fpage>27</fpage>&#x02013;<lpage>38</lpage>.</citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dugan</surname> <given-names>P.</given-names></name> <name><surname>Pourhomayoun</surname> <given-names>M.</given-names></name> <name><surname>Shiu</surname> <given-names>Y.</given-names></name> <name><surname>Paradis</surname> <given-names>R.</given-names></name> <name><surname>Rice</surname> <given-names>A.</given-names></name> <name><surname>Clark</surname> <given-names>C.</given-names></name></person-group> (<year>2013</year>). <article-title>Using high performance computing to explore large complex bioacoustic soundscapes: case study for Right Whale acoustics</article-title>. <source>Proc. Comput. Sci.</source> <volume>20</volume>, <fpage>156</fpage>&#x02013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.1016/j.procs.2013.09.254</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gilbert</surname> <given-names>J. A.</given-names></name> <name><surname>Dick</surname> <given-names>G. J.</given-names></name> <name><surname>Jenkins</surname> <given-names>B.</given-names></name> <name><surname>Heidelberg</surname> <given-names>J.</given-names></name> <name><surname>Allen</surname> <given-names>E.</given-names></name> <name><surname>Mackey</surname> <given-names>K. R.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Meeting Report: ocean &#x0201C;omics&#x0201D; science, technology, and cyberinfrastructure: current challenges and future requirements (August 20-23, 2013)</article-title>. <source>Stand. Gen. Sci.</source> <volume>9</volume>, <fpage>1251</fpage>&#x02013;<lpage>1258</lpage>. <pub-id pub-id-type="doi">10.4056/sigs.5749944</pub-id><pub-id pub-id-type="pmid">25197495</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Gray</surname> <given-names>J.</given-names></name></person-group> (<year>2009</year>). <article-title>Jim Gray and eScience: a transformed scientific method</article-title>, in <source>The Fourth Paradigm: Data-Intensive Science</source>, eds <person-group person-group-type="editor"><name><surname>Hey</surname> <given-names>T.</given-names></name> <name><surname>Tansley</surname> <given-names>S.</given-names></name> <name><surname>Tolle</surname> <given-names>K.</given-names></name></person-group> (<publisher-loc>Redmond, WA</publisher-loc>: <publisher-name>Microsoft Research</publisher-name>), <fpage>19</fpage>&#x02013;<lpage>33</lpage>.</citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Greer</surname> <given-names>A. T.</given-names></name> <name><surname>Woodson</surname> <given-names>C. B.</given-names></name> <name><surname>Smith</surname> <given-names>C. E.</given-names></name> <name><surname>Guigand</surname> <given-names>C. M.</given-names></name> <name><surname>Cowen</surname> <given-names>R. K.</given-names></name></person-group> (<year>2016</year>). <article-title>Examining mesozooplankton patch structure and its implications for trophic interactions in the northern Gulf of Mexico</article-title>. <source>J. Plank. Res.</source> <volume>38</volume>, <fpage>1115</fpage>&#x02013;<lpage>1134</lpage>. <pub-id pub-id-type="doi">10.1093/plankt/fbw033</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Greer</surname> <given-names>A. T.</given-names></name> <name><surname>Cowen</surname> <given-names>R. K.</given-names></name> <name><surname>Guigand</surname> <given-names>C. M.</given-names></name> <name><surname>Hare</surname> <given-names>J. A.</given-names></name></person-group> (<year>2015</year>). <article-title>Fine-scale planktonic habitat partitioning at a shelf-slope front revealed by a high-resolution imaging system</article-title>. <source>J. Mar. Syst.</source> <volume>142</volume>, <fpage>111</fpage>&#x02013;<lpage>125</lpage>. <pub-id pub-id-type="doi">10.1016/j.jmarsys.2014.10.008</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Greer</surname> <given-names>A. T.</given-names></name> <name><surname>Cowen</surname> <given-names>R. K.</given-names></name> <name><surname>Guigand</surname> <given-names>C. M.</given-names></name> <name><surname>Tang</surname> <given-names>D.</given-names></name></person-group> (<year>2014</year>). <article-title>The role of internal waves in larval fish interactions with potential predators and prey</article-title>. <source>Prog. Oceanogr.</source> <volume>127</volume>, <fpage>47</fpage>&#x02013;<lpage>61</lpage>. <pub-id pub-id-type="doi">10.1016/j.pocean.2014.05.010</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Greer</surname> <given-names>A. T.</given-names></name> <name><surname>Cowen</surname> <given-names>R. K.</given-names></name> <name><surname>Guigand</surname> <given-names>C. M.</given-names></name> <name><surname>Mcmanus</surname> <given-names>M. A.</given-names></name> <name><surname>Sevadjian</surname> <given-names>J. C.</given-names></name> <name><surname>Timmerman</surname> <given-names>A. H. V.</given-names></name></person-group> (<year>2013</year>). <article-title>Relationships between phytoplankton thin layers and the fine-scale vertical distributions of two trophic levels of zooplankton</article-title>. <source>J. Plankton Res.</source> <volume>35</volume>, <fpage>939</fpage>&#x02013;<lpage>956</lpage>. <pub-id pub-id-type="doi">10.1093/plankt/fbt056</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kelling</surname> <given-names>S.</given-names></name> <name><surname>Hochachka</surname> <given-names>W. M.</given-names></name> <name><surname>Fink</surname> <given-names>D.</given-names></name> <name><surname>Riedewald</surname> <given-names>M.</given-names></name> <name><surname>Caruana</surname> <given-names>R.</given-names></name> <name><surname>Ballard</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Data-intensive science: a new paradigm for biodiversity studies</article-title>. <source>Bioscience</source> <volume>59</volume>, <fpage>613</fpage>&#x02013;<lpage>620</lpage>. <pub-id pub-id-type="doi">10.1525/bio.2009.59.7.12</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kelling</surname> <given-names>S.</given-names></name> <name><surname>Lagoze</surname> <given-names>C.</given-names></name> <name><surname>Wong</surname> <given-names>W. K.</given-names></name> <name><surname>Gerbracht</surname> <given-names>J.</given-names></name> <name><surname>Fink</surname> <given-names>D.</given-names></name> <name><surname>Yu</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>eBird: a human/computer learning network to improve biodiversity conservation and research</article-title>. <source>AI Mag.</source> <volume>34</volume>, <fpage>10</fpage>&#x02013;<lpage>20</lpage>.</citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lintott</surname> <given-names>C. J.</given-names></name> <name><surname>Schawinski</surname> <given-names>K.</given-names></name> <name><surname>Slosar</surname> <given-names>A.</given-names></name> <name><surname>Land</surname> <given-names>K.</given-names></name> <name><surname>Bamford</surname> <given-names>S.</given-names></name> <name><surname>Thomas</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2008</year>). <article-title>Galaxy Zoo: morphologies derived from visual inspection of galaxies from the Sloan Digital Sky Survey</article-title>. <source>Mon. Not. R. Astron. Soc.</source> <volume>389</volume>, <fpage>1179</fpage>&#x02013;<lpage>1189</lpage>. <pub-id pub-id-type="doi">10.1111/j.1365-2966.2008.13689.x</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lukyanenko</surname> <given-names>R.</given-names></name> <name><surname>Parsons</surname> <given-names>J.</given-names></name> <name><surname>Wiersma</surname> <given-names>Y. F.</given-names></name></person-group> (<year>2014</year>). <article-title>The IQ of the crowd: understanding and improving information quality in structured user-generated content</article-title>. <source>Inform. Syst. Res.</source> <volume>25</volume>, <fpage>669</fpage>&#x02013;<lpage>689</lpage> <pub-id pub-id-type="doi">10.1287/isre.2014.0537</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Luo</surname> <given-names>J. Y.</given-names></name> <name><surname>Grassian</surname> <given-names>B.</given-names></name> <name><surname>Tang</surname> <given-names>D.</given-names></name> <name><surname>Irisson</surname> <given-names>J.-O.</given-names></name> <name><surname>Greer</surname> <given-names>A. T.</given-names></name> <name><surname>Guigand</surname> <given-names>C. M.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Environmental drivers of the fine-scale distribution of a gelatinous zooplankton community across a mesoscale front</article-title>. <source>Mar. Ecol. Prog. Ser.</source> <volume>510</volume>, <fpage>129</fpage>&#x02013;<lpage>149</lpage>. <pub-id pub-id-type="doi">10.3354/meps10908</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>McClatchie</surname> <given-names>S.</given-names></name> <name><surname>Cowen</surname> <given-names>R.</given-names></name> <name><surname>Nieto</surname> <given-names>K.</given-names></name> <name><surname>Greer</surname> <given-names>A.</given-names></name> <name><surname>Luo</surname> <given-names>J. Y.</given-names></name> <name><surname>Guigand</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Resolution of fine biological structure including small narcomedusae across a front in the Southern California Bight</article-title>. <source>J. Geophys. Res. Oceans</source> <volume>117</volume>, <fpage>C04020</fpage>. <pub-id pub-id-type="doi">10.1029/2011JC007565</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Michener</surname> <given-names>W. K.</given-names></name> <name><surname>Jones</surname> <given-names>M. B.</given-names></name></person-group> (<year>2012</year>). <article-title>Ecoinformatics: supporting ecology as a data-intensive science</article-title>. <source>Trends Ecol. Evol.</source> <volume>27</volume>, <fpage>85</fpage>&#x02013;<lpage>93</lpage>. <pub-id pub-id-type="doi">10.1016/j.tree.2011.11.016</pub-id><pub-id pub-id-type="pmid">22240191</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Millie</surname> <given-names>D. F.</given-names></name> <name><surname>Weckman</surname> <given-names>G. R.</given-names></name> <name><surname>Young</surname> <given-names>W. A.</given-names></name> <name><surname>Ivey</surname> <given-names>J. E.</given-names></name> <name><surname>Fries</surname> <given-names>D. P.</given-names></name> <name><surname>Ardjmand</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Coastal &#x0201C;Big Data&#x0201D; and nature-inspired computation: prediction potentials, uncertainties, and knowledge derivation of neural networks for an algal metric</article-title>. <source>Estuar. Coast. Shelf Sci.</source> <volume>125</volume>, <fpage>57</fpage>&#x02013;<lpage>67</lpage>. <pub-id pub-id-type="doi">10.1016/j.ecss.2013.04.001</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><collab>NASA Goddard Space Flight Center Ocean Biology Processing Group</collab></person-group> (<year>2014</year>). <source>Sea-viewing Wide Field-of-view Sensor (SeaWiFS) Ocean Color Data, NASA OB.DAAC.</source> <publisher-loc>Greenbelt, MD, USA</publisher-loc>. <publisher-name>NASA Ocean Biology Distibuted Active Archive Center</publisher-name> (OB.DAAC), Goddard Space Flight Center, Greenbelt, MD (Accessed October 25, 2015). <pub-id pub-id-type="doi">10.5067/ORBVIEW-2/SEAWIFS_OC.2014.0</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pattengill-Semmens</surname> <given-names>C. V.</given-names></name> <name><surname>Semmens</surname> <given-names>B. X.</given-names></name></person-group> (<year>2003</year>). <article-title>Conservation and management applications of the REEF volunteer fish monitoring program</article-title>. <source>Environ. Monit. Assess.</source> <volume>81</volume>, <fpage>43</fpage>&#x02013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.1023/A:1021300302208</pub-id><pub-id pub-id-type="pmid">12620003</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Peters</surname> <given-names>D. P. C.</given-names></name> <name><surname>Havstad</surname> <given-names>K. M.</given-names></name> <name><surname>Cushing</surname> <given-names>J.</given-names></name> <name><surname>Tweedie</surname> <given-names>C.</given-names></name> <name><surname>Fuentes</surname> <given-names>O.</given-names></name> <name><surname>Villanueva-Rosales</surname> <given-names>N.</given-names></name></person-group> (<year>2014</year>). <article-title>Harnessing the power of big data: infusing the scientific method with machine learning to transform ecology</article-title>. <source>Ecosphere</source> <volume>5</volume>, <fpage>1</fpage>&#x02013;<lpage>15</lpage>. <pub-id pub-id-type="doi">10.1890/ES13-00359.1</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Picheral</surname> <given-names>M.</given-names></name> <name><surname>Guidi</surname> <given-names>L.</given-names></name> <name><surname>Stemmann</surname> <given-names>L.</given-names></name> <name><surname>Karl</surname> <given-names>D. M.</given-names></name> <name><surname>Iddaoud</surname> <given-names>G.</given-names></name> <name><surname>Gorsky</surname> <given-names>G.</given-names></name></person-group> (<year>2010</year>). <article-title>The Underwater Vision Profiler 5: an advanced instrument for high spatial resolution studies of particle size spectra and zooplankton</article-title>. <source>Limnol. Oceanogr. Methods</source> <volume>8</volume>, <fpage>462</fpage>&#x02013;<lpage>473</lpage>. <pub-id pub-id-type="doi">10.4319/lom.2010.8.462</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Porter</surname> <given-names>J. H.</given-names></name> <name><surname>Hanson</surname> <given-names>P. C.</given-names></name> <name><surname>Lin</surname> <given-names>C. C.</given-names></name></person-group> (<year>2012</year>). <article-title>Staying afloat in the sensor data deluge</article-title>. <source>Trends Ecol. Evol.</source> <volume>27</volume>, <fpage>121</fpage>&#x02013;<lpage>129</lpage>. <pub-id pub-id-type="doi">10.1016/j.tree.2011.11.009</pub-id><pub-id pub-id-type="pmid">22206661</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sedberry</surname> <given-names>G. R.</given-names></name> <name><surname>Fautin</surname> <given-names>D. G.</given-names></name> <name><surname>Feldman</surname> <given-names>M.</given-names></name> <name><surname>Fornwall</surname> <given-names>M. D.</given-names></name> <name><surname>Goldstein</surname> <given-names>P.</given-names></name> <name><surname>Guralnick</surname> <given-names>R. P.</given-names></name></person-group> (<year>2011</year>). <article-title>OBIS-USA: a data-sharing legacy of census of marine life</article-title>. <source>Oceanography</source> <volume>24</volume>, <fpage>166</fpage>&#x02013;<lpage>173</lpage>. <pub-id pub-id-type="doi">10.5670/oceanog.2011.36</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shamir</surname> <given-names>L.</given-names></name> <name><surname>Yerby</surname> <given-names>C.</given-names></name> <name><surname>Simpson</surname> <given-names>R.</given-names></name> <name><surname>von Benda-Beckmann</surname> <given-names>A. M.</given-names></name> <name><surname>Tyack</surname> <given-names>P.</given-names></name> <name><surname>Samarra</surname> <given-names>F.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Classification of large acoustic datasets using machine learning and crowdsourcing &#x02013; application to whale calls</article-title>. <source>J. Acoust. Soc. Am.</source> <volume>135</volume>, <fpage>953</fpage>&#x02013;<lpage>962</lpage>. <pub-id pub-id-type="doi">10.1121/1.4861348</pub-id><pub-id pub-id-type="pmid">25234903</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>S. D.</given-names></name> <name><surname>Edgar</surname> <given-names>R. J.</given-names></name></person-group> (<year>2014</year>). <article-title>Documenting the density of subtidal marine debris across multiple marine and coastal habitats</article-title>. <source>PLoS ONE</source> <volume>9</volume>:<fpage>e94593</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0094593</pub-id><pub-id pub-id-type="pmid">24743690</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Thomson</surname> <given-names>C. W.</given-names></name> <name><surname>Murray</surname> <given-names>J.</given-names></name></person-group> (<year>1895</year>). <source>Voyage of the H.M.S. Challenger during the Years 1875-76: A Summary of the Scientific Reports. First Part</source>. Digital edition prepared by Brossard, D.C. (2013); <publisher-name>Eyre &#x00026; Spottiswoode</publisher-name>, <publisher-loc>London</publisher-loc>, <fpage>107</fpage>&#x02013;<lpage>1247</lpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="http://www.19thcenturyscience.org/HMSC/HMSC-Reports/1895-Summary/htm/doc.html">http://www.19thcenturyscience.org/HMSC/HMSC-Reports/1895-Summary/htm/doc.html</ext-link></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vezzulli</surname> <given-names>L.</given-names></name> <name><surname>Reid</surname> <given-names>P. C.</given-names></name></person-group> (<year>2003</year>). <article-title>The CPR survey (1948-1997): a gridded database browser of plankton abundance in the North Sea</article-title>. <source>Prog. Oceanogr.</source> <volume>58</volume>, <fpage>327</fpage>&#x02013;<lpage>336</lpage>. <pub-id pub-id-type="doi">10.1016/j.pocean.2003.08.011</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>R. Y.</given-names></name> <name><surname>Strong</surname> <given-names>D. M.</given-names></name></person-group> (<year>1996</year>). <article-title>What data quality means to data consumers</article-title>. <source>J. Manag. Inform. Syst.</source> <volume>12</volume>, <fpage>5</fpage>&#x02013;<lpage>33</lpage>.</citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wiley</surname> <given-names>E. O.</given-names></name> <name><surname>McNyset</surname> <given-names>K. M.</given-names></name> <name><surname>Peterson</surname> <given-names>A. T.</given-names></name> <name><surname>Stewart</surname> <given-names>A.</given-names></name></person-group> (<year>2003</year>). <article-title>Niche modeling and geographic range predictions in the marine environment using a machine-learning algorithm</article-title>. <source>Oceanography</source> <volume>16</volume>, <fpage>120</fpage>&#x02013;<lpage>127</lpage>. <pub-id pub-id-type="doi">10.5670/oceanog.2003.42</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Woolley</surname> <given-names>A. W.</given-names></name> <name><surname>Chabris</surname> <given-names>C. F.</given-names></name> <name><surname>Pentland</surname> <given-names>A.</given-names></name> <name><surname>Hashmi</surname> <given-names>N.</given-names></name> <name><surname>Malone</surname> <given-names>T. W.</given-names></name></person-group> (<year>2010</year>). <article-title>Evidence for a collective intelligence factor in the performance of human groups</article-title>. <source>Science</source> <volume>330</volume>, <fpage>686</fpage>&#x02013;<lpage>688</lpage>. <pub-id pub-id-type="doi">10.1126/science.1193147</pub-id><pub-id pub-id-type="pmid">20929725</pub-id></citation></ref>
</ref-list>
</back>
</article>