<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xml:lang="EN" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Plant Sci.</journal-id>
<journal-title>Frontiers in Plant Science</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Plant Sci.</abbrev-journal-title>
<issn pub-type="epub">1664-462X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fpls.2021.787127</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Plant Science</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>The Herbarium 2021 Half&#x02013;Earth Challenge Dataset and Machine Learning Competition</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>de Lutio</surname> <given-names>Riccardo</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1425502/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Park</surname> <given-names>John Y.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1568935/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Watson</surname> <given-names>Kimberly A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1582091/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>D&#x00027;Aronco</surname> <given-names>Stefano</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Wegner</surname> <given-names>Jan D.</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Wieringa</surname> <given-names>Jan J.</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Tulig</surname> <given-names>Melissa</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1580297/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Pyle</surname> <given-names>Richard L.</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/405696/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Gallaher</surname> <given-names>Timothy J.</given-names></name>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Brown</surname> <given-names>Gillian</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Guymer</surname> <given-names>Gordon</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Franks</surname> <given-names>Andrew</given-names></name>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Ranatunga</surname> <given-names>Dhahara</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Baba</surname> <given-names>Yumiko</given-names></name>
<xref ref-type="aff" rid="aff7"><sup>7</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1551452/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Belongie</surname> <given-names>Serge J.</given-names></name>
<xref ref-type="aff" rid="aff8"><sup>8</sup></xref>
</contrib>
<contrib contrib-type="author">
<name><surname>Michelangeli</surname> <given-names>Fabi&#x000E1;n A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/486063/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ambrose</surname> <given-names>Barbara A.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/28043/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Little</surname> <given-names>Damon P.</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/1551187/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>EcoVision Lab, Department of Civil, Environmental and Geomatic Engineering, ETH Z&#x000FC;rich</institution>, <addr-line>Zurich</addr-line>, <country>Switzerland</country></aff>
<aff id="aff2"><sup>2</sup><institution>New York Botanical Garden</institution>, <addr-line>Bronx, NY</addr-line>, <country>United States</country></aff>
<aff id="aff3"><sup>3</sup><institution>Faculty of Science, Institute for Computational Science, University of Zurich</institution>, <addr-line>Zurich</addr-line>, <country>Switzerland</country></aff>
<aff id="aff4"><sup>4</sup><institution>Naturalis Biodiversity Center</institution>, <addr-line>Leiden</addr-line>, <country>Netherlands</country></aff>
<aff id="aff5"><sup>5</sup><institution>Bishop Museum</institution>, <addr-line>Honolulu, HI</addr-line>, <country>United States</country></aff>
<aff id="aff6"><sup>6</sup><institution>Queensland Herbarium, Department of Environment and Science</institution>, <addr-line>Toowong, QLD</addr-line>, <country>Australia</country></aff>
<aff id="aff7"><sup>7</sup><institution>Auckland War Memorial Museum T&#x00101;maki Paenga Hira</institution>, <addr-line>Auckland</addr-line>, <country>New Zealand</country></aff>
<aff id="aff8"><sup>8</sup><institution>Department of Computer Science, University of Copenhagen, and Pioneer Centre for AI</institution>, <addr-line>Copenhagen</addr-line>, <country>Denmark</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Alexis Joly, Research Centre Inria Sophia Antipolis M&#x000E9;diterran&#x000E9;e, France</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Gregory W. Stull, Kunming Institute of Botany, CAS, China; Emily Bellis, Arkansas State University, United States</p></fn>
<corresp id="c001">&#x0002A;Correspondence: Riccardo de Lutio <email>rdelutio&#x00040;ethz.ch</email></corresp>
<fn fn-type="other" id="fn001"><p>This article was submitted to Technical Advances in Plant Science, a section of the journal Frontiers in Plant Science</p></fn></author-notes>
<pub-date pub-type="epub">
<day>01</day>
<month>02</month>
<year>2022</year>
</pub-date>
<pub-date pub-type="collection">
<year>2021</year>
</pub-date>
<volume>12</volume>
<elocation-id>787127</elocation-id>
<history>
<date date-type="received">
<day>30</day>
<month>09</month>
<year>2021</year>
</date>
<date date-type="accepted">
<day>20</day>
<month>12</month>
<year>2021</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2022 de Lutio, Park, Watson, D&#x00027;Aronco, Wegner, Wieringa, Tulig, Pyle, Gallaher, Brown, Guymer, Franks, Ranatunga, Baba, Belongie, Michelangeli, Ambrose and Little.</copyright-statement>
<copyright-year>2022</copyright-year>
<copyright-holder>de Lutio, Park, Watson, D&#x00027;Aronco, Wegner, Wieringa, Tulig, Pyle, Gallaher, Brown, Guymer, Franks, Ranatunga, Baba, Belongie, Michelangeli, Ambrose and Little</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract>
<p>Herbarium sheets present a unique view of the world&#x00027;s botanical history, evolution, and biodiversity. This makes them an all&#x02013;important data source for botanical research. With the increased digitization of herbaria worldwide and advances in the domain of fine&#x02013;grained visual classification which can facilitate automatic identification of herbarium specimen images, there are many opportunities for supporting and expanding research in this field. However, existing datasets are either too small, or not diverse enough, in terms of represented taxa, geographic distribution, and imaging protocols. Furthermore, aggregating datasets is difficult as taxa are recognized under a multitude of names and must be aligned to a common reference. We introduce the Herbarium 2021 Half&#x02013;Earth dataset: the largest and most diverse dataset of herbarium specimen images, to date, for automatic taxon recognition. We also present the results of the Herbarium 2021 Half&#x02013;Earth challenge, a competition that was part of the Eighth Workshop on Fine-Grained Visual Categorization (FGVC8) and hosted by Kaggle to encourage the development of models to automatically identify taxa from herbarium sheet images.</p>
</abstract>
<kwd-group>
<kwd>herbarium specimen image</kwd>
<kwd>fine-grained visual categorization</kwd>
<kwd>machine learning competition</kwd>
<kwd>hierarchical classification</kwd>
<kwd>datasets</kwd>
</kwd-group>
<counts>
<fig-count count="10"/>
<table-count count="3"/>
<equation-count count="1"/>
<ref-count count="63"/>
<page-count count="15"/>
<word-count count="9010"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>1. Introduction</title>
<p>Herbaria, like other natural history collections, are immense primary data repositories documenting biodiversity across space and time over the last 500 years (Stefanaki et al., <xref ref-type="bibr" rid="B46">2019</xref>). Each specimen contains a wealth of information including geographic occurrence data, phenotype, genotype, phenological status, and biotic interactions (Funk, <xref ref-type="bibr" rid="B15">2003</xref>; Heberling and Burke, <xref ref-type="bibr" rid="B18">2019</xref>). Collectively herbarium specimens are analyzed for studies in taxonomy, systematics, floristics, ecology, phenology, conservation, and global environmental change (Funk, <xref ref-type="bibr" rid="B15">2003</xref>; Calinger et al., <xref ref-type="bibr" rid="B5">2013</xref>; Willis et al., <xref ref-type="bibr" rid="B58">2017</xref>; Lang et al., <xref ref-type="bibr" rid="B31">2019</xref>; Albani Rocchetti et al., <xref ref-type="bibr" rid="B1">2021</xref>).</p>
<p>Worldwide efforts to digitize and electronically mobilize biodiversity data for the estimated 396 million herbarium specimens, housed in 3,400 herbaria (Thiers, <xref ref-type="bibr" rid="B50">2021</xref>), have greatly amplified their use in research (Heberling et al., <xref ref-type="bibr" rid="B20">2019</xref>; Nelson and Ellis, <xref ref-type="bibr" rid="B36">2019</xref>), including projects to understand, predict, and ameliorate increasing environmental threats to biodiversity (Intergovernmental Science&#x02013;Policy Platform on Biodiversity and Ecosystem Services, <xref ref-type="bibr" rid="B25">2019</xref>; Lang et al., <xref ref-type="bibr" rid="B31">2019</xref>). Plants are essential to life on Earth, yet an estimated 37&#x02013;44% of all vascular plant species are threatened with extinction (Nic Lughadha et al., <xref ref-type="bibr" rid="B37">2020</xref>), underscoring the urgency to identify and classify the estimated 70,000 flowering plant species not yet described (Bebber et al., <xref ref-type="bibr" rid="B2">2010</xref>; Joppa et al., <xref ref-type="bibr" rid="B28">2011</xref>). Half of these new species are predicted to be already preserved in herbaria, awaiting an average of 35 years for detection and description from the date of first specimen collection (Bebber et al., <xref ref-type="bibr" rid="B2">2010</xref>). Contributing to this delay is the dwindling number of taxonomists with broad plant identification skills to recognize new species, who are under ever increasing demands on their time and expertise (Secretariat of the Convention on Biological Diversity, <xref ref-type="bibr" rid="B45">2007</xref>).</p>
<p>Recent advances in machine learning and computer vision as well as increased biodiversity data mobilization through global data aggregators, such as the Global Biodiversity Information Facility (GBIF), enable the development of models to address a variety of plant&#x02013;science&#x02013;related questions and potentially overcome such &#x0201C;taxonomic impediments&#x0201D; (Secretariat of the Convention on Biological Diversity, <xref ref-type="bibr" rid="B45">2007</xref>; Heberling et al., <xref ref-type="bibr" rid="B19">2021</xref>). For example, the automatic identification of specimens has shown particularly promising results from learning&#x02013;based approaches (review by W&#x000E4;ldchen and M&#x000E4;der, <xref ref-type="bibr" rid="B53">2018</xref>). Many studies have focused on small sets of closely&#x02013;related plant taxa (Clark et al., <xref ref-type="bibr" rid="B8">2012</xref>; Nasir et al., <xref ref-type="bibr" rid="B35">2014</xref>; Unger et al., <xref ref-type="bibr" rid="B52">2016</xref>; Kho et al., <xref ref-type="bibr" rid="B29">2017</xref>; Schuettpelz et al., <xref ref-type="bibr" rid="B44">2017</xref>; Pryer et al., <xref ref-type="bibr" rid="B40">2020</xref>) whereas others tackle the more challenging problem of automatic identification of a large number of taxa (Carranza-Rojas et al., <xref ref-type="bibr" rid="B7">2017</xref>; Younis et al., <xref ref-type="bibr" rid="B61">2018</xref>; Little et al., <xref ref-type="bibr" rid="B33">2020</xref>). Many automatic identification studies focus on recognition from leaves alone (Wijesingha and Marikar, <xref ref-type="bibr" rid="B56">2012</xref>; Nasir et al., <xref ref-type="bibr" rid="B35">2014</xref>; Unger et al., <xref ref-type="bibr" rid="B52">2016</xref>; Wilf et al., <xref ref-type="bibr" rid="B57">2016</xref>; Kho et al., <xref ref-type="bibr" rid="B29">2017</xref>). Similar techniques have also been used for phenological studies and trait recognition (Clark et al., <xref ref-type="bibr" rid="B8">2012</xref>; Ubbens and Stavness, <xref ref-type="bibr" rid="B51">2017</xref>; Younis et al., <xref ref-type="bibr" rid="B61">2018</xref>; Lorieul et al., <xref ref-type="bibr" rid="B34">2019</xref>; Brenskelle et al., <xref ref-type="bibr" rid="B3">2020</xref>; Davis et al., <xref ref-type="bibr" rid="B10">2020</xref>; Go&#x000EB;au et al., <xref ref-type="bibr" rid="B16">2020</xref>; Pearson et al., <xref ref-type="bibr" rid="B39">2020</xref>; Pryer et al., <xref ref-type="bibr" rid="B40">2020</xref>).</p>
<p>Citizen science initiatives, such as iNaturalist (Horn et al., <xref ref-type="bibr" rid="B22">2018</xref>), Pl&#x00040;ntNet (Joly et al., <xref ref-type="bibr" rid="B27">2016</xref>), and ObsIdentify (Hogeweg et al., <xref ref-type="bibr" rid="B21">2019</xref>), have popularized species recognition as a challenging real&#x02013;world classification task among the computer vision community. They are particularly popular because of the size as well as the imbalanced and fine&#x02013;grained nature of their respective datasets. Through a series of online algorithm competitions (e.g., Horn et al., <xref ref-type="bibr" rid="B22">2018</xref>; Little et al., <xref ref-type="bibr" rid="B33">2020</xref>), automated identification techniques have become increasingly accurate.</p>
<p>Existing digitized herbarium specimen datasets designed for computer vision approaches present some limitations: they are either small, targeted at specific taxa, representative of only a small geographic region, or contain images produced using only one imaging protocol (generally institution specific; <xref ref-type="table" rid="T1">Table 1</xref>). In this paper we introduce the Herbarium 2021 Half&#x02013;Earth dataset, which aims to address the limitations aforementioned and is the largest and most diverse dataset of herbarium specimen images for automatic taxon recognition to date. We also present the results from the challenge of the same name: the Herbarium 2021 Half&#x02013;Earth challenge, a competition that was organized as part of the 8<italic>th</italic> workshop for Fine&#x02013;Grained Visual Categorization at the Computer Vision and Pattern Recognition conference (CVPR) in 2021. The competition was hosted on Kaggle<xref ref-type="fn" rid="fn0001"><sup>1</sup></xref> and took place between March 10<italic>th</italic> and May 27<italic>th</italic> 2021.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Summary of existing herbarium sheet image datasets.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Dataset</bold></th>
<th valign="top" align="center"><bold>Images</bold></th>
<th valign="top" align="center"><bold>Taxa</bold></th>
<th valign="top" align="center"><bold>Vascular plant representation (%)</bold></th>
<th valign="top" align="center"><bold>Institutions</bold></th>
<th valign="top" align="left"><bold>Geographic range</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Dillen et al., <xref ref-type="bibr" rid="B12">2019</xref></td>
<td valign="top" align="center">1,900</td>
<td valign="top" align="center">1,580</td>
<td valign="top" align="center">0.34</td>
<td valign="top" align="center">9</td>
<td valign="top" align="left">Global</td>
</tr>
<tr>
<td valign="top" align="left">Herbarium 255 (Carranza-Rojas et al., <xref ref-type="bibr" rid="B7">2017</xref>)</td>
<td valign="top" align="center">11,071</td>
<td valign="top" align="center">255</td>
<td valign="top" align="center">0.05</td>
<td valign="top" align="center">1</td>
<td valign="top" align="left">Costa Rica</td>
</tr>
<tr>
<td valign="top" align="left">Herbarium 1K (Carranza-Rojas et al., <xref ref-type="bibr" rid="B7">2017</xref>)</td>
<td valign="top" align="center">253,733</td>
<td valign="top" align="center">1,204</td>
<td valign="top" align="center">0.26</td>
<td valign="top" align="center">1</td>
<td valign="top" align="left">France</td>
</tr>
<tr>
<td valign="top" align="left">Herbarium 2019 (Tan et al., <xref ref-type="bibr" rid="B48">2019</xref>)</td>
<td valign="top" align="center">46,000</td>
<td valign="top" align="center">680</td>
<td valign="top" align="center">0.15</td>
<td valign="top" align="center">1</td>
<td valign="top" align="left">Americas</td>
</tr>
<tr>
<td valign="top" align="left">Herbarium 2020</td>
<td valign="top" align="center">1,169,039</td>
<td valign="top" align="center">32,094</td>
<td valign="top" align="center">6.85</td>
<td valign="top" align="center">1</td>
<td valign="top" align="left">Americas</td>
</tr>
<tr>
<td valign="top" align="left">Herbarium 2021</td>
<td valign="top" align="center">2,500,779</td>
<td valign="top" align="center">64,500</td>
<td valign="top" align="center">13.76</td>
<td valign="top" align="center">5</td>
<td valign="top" align="left">Americas, Oceania, and Pacific</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>The percent of vascular plant taxa represented is based on the 468,759 LCVP (Freiberg et al., <xref ref-type="bibr" rid="B14">2020</xref>) &#x0201C;accepted&#x0201D; and &#x0201C;unresolved&#x0201D; taxa. Because different taxonomies were used as standards for the various datasets, the reported percentage can only be considered an approximation. Note that the Herbarium 2019 dataset focuses on the flowering plant family Melastomataceae, while the other datasets include representatives across vascular plants</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>The goal of the competition was to encourage the development of models to automatically identify a very large number of taxa from herbarium sheet images, and evaluate which deep learning approaches have the best performance in this setting. This is the third iteration of the Herbarium challenge: the Herbarium 2019 challenge (Tan et al., <xref ref-type="bibr" rid="B48">2019</xref>; Little et al., <xref ref-type="bibr" rid="B33">2020</xref>) focused on the flowering plant family Melastomataceae and contained 46,469 digitally imaged herbarium specimens representing 683 species. The Melastomataceae is a large family with 166 recognized genera and 5,892 species (Freiberg et al., <xref ref-type="bibr" rid="B14">2020</xref>). The Herbarium 2020 dataset contained 1,169,039 images representing 32,094 plant species. This challenge focused on vascular land plants of the Americas. Compared to the previous datasets the 2021 Half&#x02013;Earth dataset is larger in terms of both number of taxa, and number of images, with 2,500,779 images and 64,500 taxa. After introducing the dataset and presenting the results of the competition, we discuss possible outlooks in order to leverage the full potential of deep learning models and herbarium data.</p>
</sec>
<sec sec-type="methods" id="s2">
<title>2. Methods</title>
<sec>
<title>2.1. The Herbarium 2021 Half&#x02013;Earth Dataset</title>
<p>The Herbarium 2021 Half&#x02013;Earth dataset<xref ref-type="fn" rid="fn0002"><sup>2</sup></xref> includes more than 2.5 million images of vascular plant specimens (including lycophytes, ferns, gymnosperms, and flowering plants) representing 64,500 taxa from the Americas, Oceania, and Pacific<xref ref-type="fn" rid="fn0003"><sup>3</sup></xref>. The images are provided by the New York Botanical Garden (NY), Bishop Museum (BPBM), Naturalis Biodiversity Center (NL), Queensland Herbarium (BRI), and Auckland War Memorial Museum (AK). The most exact labels are, in many cases, infraspecific (subspecies, varieties, forms, etc.) or nothospecies (hybrids), neither of which can be characterized as &#x0201C;species&#x0201D;, thus the terms &#x0201C;taxon&#x0201D; and &#x0201C;taxa&#x0201D; are used as generic descriptors of taxonomic labels. In addition to labels for species&#x02013;level and below, labels at higher levels in the taxonomic hierarchy are also included: family and order. This allows for experimentation with methods that address label hierarchy and label similarity. These labels may also be supplemented by more fine&#x02013;grained estimates of difference among taxa available from other sources (e.g., Jin and Qian, <xref ref-type="bibr" rid="B26">2019</xref>). The dataset is characterized by a skewed long tail distribution (<xref ref-type="fig" rid="F1">Figure 1</xref>). Whereas some taxa can be represented by more than 1,000 images, other taxa have only three images. This dataset includes only images of vascular plants&#x02014;the group of plants that includes lycophytes, ferns, gymnosperms, and flowering plants (<xref ref-type="fig" rid="F2">Figure 2</xref>).</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Distribution of training images per taxon. The Herbarium 2021 Half&#x02013;Earth dataset is highly imbalanced. Featured taxa are from top to bottom: <italic>Ericameria nauseosa</italic> (Pall. ex Pursh) G.L. Nesom &#x00026; G.I. Baird (Asteraceae), <italic>Bidens sulphurea</italic> (Cav.) Sch. Bip. (Asteraceae), and <italic>Solanum rixosum</italic> A.R. Bean (Solanaceae). Taxon names are usually followed by name of person(s) first formally describing the taxon in the scientific literature. Here, higher level hierarchy of each taxon is followed by family name in parentheses.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0001.tif"/>
</fig>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Example of images in the Herbarium 2021 Half&#x02013;Earth dataset.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0002.tif"/>
</fig>
<sec>
<title>2.1.1. Dataset Challenges</title>
<p>The Herbarium 2021 Half&#x02013;Earth dataset is challenging for multiple reasons. First, of course, the large imbalance (<xref ref-type="fig" rid="F1">Figure 1</xref>): the imbalance factor (ratio of the number of images for the most represented class to the number of images for the least represented class) for the dataset is 1,654.5. Second, the variation within species is high: herbarium specimens capture plants at different growth&#x02013;stages (e.g., juvenile vs. adult), with different sets of plant parts (e.g., leaves and flowers vs. leaves and fruit; <xref ref-type="fig" rid="F3">Figure 3</xref>) or simply different individuals can present different visual appearances. In addition, the techniques used to press, dry, and mount specimens vary among collectors and collecting expeditions&#x02014;these differences can change the appearance of specimens dramatically (e.g., collecting in alcohol often causes leaves to turn black). Arbitrary aesthetic decisions made while processing specimens can result in specimens that differ dramatically in appearance even though they are simply different parts of the same individual plant (<xref ref-type="fig" rid="F4">Figure 4</xref>). In a herbarium collection, every attempt to conserve dried specimens is made, but in practice older specimens become more fragile and suffer damage as they age leading to some specimens being less complete and more damaged than others. Third, the visual similarity among species can be high (<xref ref-type="fig" rid="F5">Figure 5</xref>). Finally, the diagnostic morphological features that botanists use to identify species are often very small and thus require a model that is able to handle high&#x02013;resolution images and can focus on specific details (Cope et al., <xref ref-type="bibr" rid="B9">2012</xref>; W&#x000E4;ldchen and M&#x000E4;der, <xref ref-type="bibr" rid="B53">2018</xref>).</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Example of visually different images corresponding to the same species: <italic>Abarema brachystachya</italic> (DC.) Barneby and J. W. Grimes (Fabaceae). The observed differences are primarily due to different reproductive stages: early flowering, late flowering, and fruit.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0003.tif"/>
</fig>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Different specimens of <italic>Arbutus xalapensis</italic> Kunth (Ericaceae) made from the same individual plant at the same time by the same collector using the same pressing, drying, and mounting protocol.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0004.tif"/>
</fig>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Example of visually similar images from different <italic>Alyssum</italic> species (Brassicaceae): <italic>A. alyssoides</italic> (L.) L., <italic>A. desertorum</italic> Stapf, <italic>A. simplex</italic> Rudolphi, <italic>A. szovitsianum</italic> Fisch. and C. A. Mey.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0005.tif"/>
</fig>
</sec>
<sec>
<title>2.1.2. Data Preprocessing</title>
<p>In this section, we give an overview of how the Herbarium 2021 Half&#x02013;Earth dataset was preprocessed. <xref ref-type="fig" rid="F6">Figure 6</xref> presents some example herbarium sheet images before and after the preprocessing steps.</p>
<fig id="F6" position="float">
<label>Figure 6</label>
<caption><p>Example of images before <bold>(left)</bold> and after <bold>(right)</bold> preprocessing.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0006.tif"/>
</fig>
<sec>
<title>2.1.2.1. Label Alignment</title>
<p>Herbarium specimens of the same taxon may have been labeled in various ways due to differences in the interpretation of taxon circumscriptions, nomenclature changes, and/or errors. For example, over time <italic>Pilosella piloselloides</italic> (Vill.) Soj&#x000E1;k (Asteraceae) has been known by at least 526 different names (Freiberg et al., <xref ref-type="bibr" rid="B14">2020</xref>). To ameliorate this situation as much as possible, image labels are standardized to the Leipzig Catalogue of Vascular Plants (LCVP v1.0.2; Freiberg et al., <xref ref-type="bibr" rid="B14">2020</xref>). Labels in the dataset have an LCVP status of either &#x0201C;accepted&#x0201D; or &#x0201C;unresolved&#x0201D;. The data exported from the institutional databases were first processed to find labels that exactly matched LCVP. For labels that did not precisely match, we then searched for long unambiguous partial matches to LCVP: the label was shortened by removing the rightmost word and then searched for a match that produced only one LCVP output taxon; if no match was found, this was repeated until the label contained only two words. Labels that still did not unambiguously match LCVP, were matched using tre-agrep (Wu and Manber, <xref ref-type="bibr" rid="B59">1992</xref>) allowing an increasing amount of mismatch (10&#x02013;30% of label length; all weights were set to 1). Matches returned by tre-agrep were manually reviewed (8,430 labels passed manual review). Images with labels that could not be coerced into matching LCVP were excluded from the dataset (<italic>c</italic>. 73 thousand images).</p>
</sec>
<sec>
<title>2.1.2.2. Image Blurring</title>
<p>Herbarium specimens always have a hand&#x02013;written or printed label on the sheet (usually lower right&#x02013;hand corner), which includes information about the name of the taxon, the geographic location where it was collected, the date of collection, and the person or team of people who collected it. In addition, annotation labels are often added to the specimen to correct or update information on the original label&#x02014;these are sequentially added in the empty space above the original label. Specimens often also have institutional labels or stamps indicating the herbarium in which the specimen is archived and a barcode label corresponding to an institutional database entry. Specimens may also include field tags with identification numbers attached directly to the plant. Images usually include color and measurement scales as well as institutional logos. All of these labels can of course, help identify the specimen, thus this information in the dataset was blurred in order to force models to learn about the plants themselves rather than the label text. A pretrained EAST text detection model (Zhou et al., <xref ref-type="bibr" rid="B63">2017</xref>) was used to detect these labels. This model outputs bounding boxes around the detected text. The bounding boxes that overlapped by a sufficient margin were merged and those that were too small were filtered out. The resulting regions were then heavily blurred. First, a mean blur was applied, then a single Gaussian blur with added noise, and then a smooth alpha map to blend into the original (<xref ref-type="fig" rid="F6">Figure 6</xref>). Finally, images where more than 25% of the image was blurred were excluded from the dataset, as those represent, in most cases, wrong predictions from EAST. The text detection model was deliberately tuned to have a high specificity, in order to avoid unnecessarily blurring plant parts. Even though, this means that there are images where part of the labels are missed by the blurring algorithm.</p>
</sec>
<sec>
<title>2.1.2.3. Image Resizing</title>
<p>Herbarium sheets are digitized as very high&#x02013;resolution images to preserve as much of the detail as possible. A common image size is around 6000 &#x000D7; 4000 pixels. This is very large even for networks that are designed to work with higher resolutions. All images in the dataset are resized to a dimension of 1,000 pixels (while preserving the aspect ratio), in order to make the overall size of the dataset more accessible.</p>
</sec>
<sec>
<title>2.1.2.4. Dataset Split</title>
<p>Herbarium 2021 contains images from 64,500 taxa at the species&#x02013;level or below with 2,257,759 in the training set and 243,020 in the test set. The data has been split to obtain an approximately even number of images across taxa in the test set by capping the maximum number of images per taxon at 10. For taxa that have few images a 80%/20% split for training/test is used&#x02014;each category has a minimum of three images: at least one in the test set and two in the training set.</p>
</sec>
<sec>
<title>2.1.2.5. Hierarchical Labels</title>
<p>In addition to the name of the taxon, labels for the family and order are provided. The herbarium sheet images provided in this dataset represent 64,500 different taxa, belonging to 451 families and 81 orders. This enables the development of methods that utilize hierarchical information. Ideally, mistakes between closely related taxa should not be treated equal to mistakes between very distant taxa. See Section 2.2 for an example of a loss function that leverages hierarchical labels.</p>
</sec>
</sec>
</sec>
<sec>
<title>2.2. Baselines</title>
<p>In order to have a reference value for the dataset performance, a standard ResNet-50 (He et al., <xref ref-type="bibr" rid="B17">2016</xref>) was trained as a baseline method. A balanced sampling strategy was used to mitigate the impact of the imbalance on the classifier. The images were resized to 256 &#x000D7; 256 pixels and standard data augmentations were applied (small rotations, horizontal flips, color&#x02013;jitter, and center&#x02013;crop to 224 &#x000D7; 224 pixels). The model was initialized with weights pretrained on ImageNet (Deng et al., <xref ref-type="bibr" rid="B11">2009</xref>). Finally, the model was trained using the standard cross&#x02013;entropy loss, a batch size of 32, a stochastic gradient descent with a learning rate of 1 &#x000D7; 10<sup>&#x02212;3</sup> which is further reduced when a plateau was reached and a momentum factor of 0.9. The model was trained for a total of 10 epochs (with 70,555 batches per epoch).</p>
<p>To integrate hierarchical labels, the marginalization loss function proposed in Kumar and Zheng (<xref ref-type="bibr" rid="B30">2017</xref>) was adopted. The basic idea behind the marginalization loss is to simultaneously apply a classification loss at all the levels of the hierarchy. In order to compute the marginalization loss the label and the predicted distribution at each level of the hierarchy are needed: the label can simply be obtained by looking up the family and order; the predicted distribution for the family (or order) can be estimated from the sum of scores for all the taxa in each family (or order) in what resembles a marginalization procedure. Note that if the network predicts a distribution over the taxa, the marginalization over family and order also leads to a valid categorical distribution. A cross&#x02013;entropy loss at the taxa level as well as the family and order level of hierarchy can be applied&#x02014;this should ideally improve the regularization power of the network.</p>
</sec>
<sec>
<title>2.3. Evaluation Metrics</title>
<p>In order to evaluate the classification performance the main metric chosen for the Herbarium 2021 challenge was the F<sub>1</sub> score, which is equal to:</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mtext>F</mml:mtext></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mfrac><mml:mrow><mml:mtext>Pre&#x000A0;</mml:mtext><mml:mo>&#x000B7;</mml:mo><mml:mtext>&#x000A0;Rec</mml:mtext></mml:mrow><mml:mrow><mml:mtext>Pre&#x000A0;</mml:mtext><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;Rec</mml:mtext></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where Pre denotes the precision and Rec the recall. This score is computed for every taxon separately and then averaged across all taxa to get the final score. Accuracy Acc and mean class accuracy Mca (also know as per&#x02013;class accuracy) are also reported.</p>
<p>As an additional performance metric, the patristic distance between the expected and the predicted classes is also reported. Patristic distances were extracted from a dated genus&#x02013;level phylogeny pruned to include only the taxa in the dataset (Jin and Qian, <xref ref-type="bibr" rid="B26">2019</xref>). Within genera, distances among taxa were crudely interpolated by adding 10% of the distance between each genus and its sister genus.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>3. Results</title>
<sec>
<title>3.1. Competition Results</title>
<p>The Herbarium 2021 Half&#x02013;Earth challenge received 573 entries submitted by 108 competitors divided across 80 teams. As seen in <xref ref-type="fig" rid="F7">Figure 7</xref>, there are large gaps in performance between the competitors. Focusing on the top&#x02013;five teams of the competition: all had F<sub>1</sub> performance above 0.680 on the test set. The teams are (in order of decreasing F<sub>1</sub> scores): CIPP (0.757), HaeC (0.735), Brendan Rapazzo (0.689), Qidian213 (0.687), Undergrad &#x00026; Botany Joe (0.682; <xref ref-type="table" rid="T2">Table 2</xref>). All of the top&#x02013;five approaches used relatively high resolution images (352 &#x000D7; 352 pixels or higher). The top&#x02013;three solutions were ensembles of models, with the top&#x02013;two teams combining the predictions from different models and the third place team combining predictions made by the same model at the different stages of the training process.</p>
<fig id="F7" position="float">
<label>Figure 7</label>
<caption><p>F<sub>1</sub> scores of the 50 best performing teams.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0007.tif"/>
</fig>
<table-wrap position="float" id="T2">
<label>Table 2</label>
<caption><p>Summary of the top competitors&#x00027; solutions and performance.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Team ranking</bold></th>
<th valign="top" align="left"><bold>1st</bold></th>
<th valign="top" align="left"><bold>2nd</bold></th>
<th valign="top" align="left"><bold>3rd</bold></th>
<th valign="top" align="left"><bold>4th</bold></th>
<th valign="top" align="left"><bold>5th</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Team name (Organization)</td>
<td valign="top" align="left">CIPP (Alibaba Group)</td>
<td valign="top" align="left">HaeC (Postech)</td>
<td valign="top" align="left">Brendan Rapazzo (Cornell University)</td>
<td valign="top" align="left">Qidian213</td>
<td valign="top" align="left">Undergrad &#x00026; Botany Joe (The University of Tennessee)</td>
</tr>
<tr>
<td valign="top" align="left">Team members</td>
<td valign="top" align="left">Baoming Yan, Bo Gao, Xiao Liu, Lin Wang, and Chao Ban</td>
<td/>
<td valign="top" align="left">Brendan Rapazzo</td>
<td valign="top" align="left">&#x02014;</td>
<td valign="top" align="left">Dax Ledesma and Joey Shaw</td>
</tr>
<tr>
<td valign="top" align="left">Model architecture</td>
<td valign="top" align="left">ResNest101, ResNeXt101-IBN-a, ResNeXt101</td>
<td valign="top" align="left">TResNet-M, TResNet-M-21k, TResNet-L, GENet-L, ECA-NFNet-L0</td>
<td valign="top" align="left">SE-ResNeXt101</td>
<td valign="top" align="left">&#x02014;</td>
<td valign="top" align="left">SE-ResNeXt50</td>
</tr>
<tr>
<td valign="top" align="left">Feature extractor parameters (M)</td>
<td valign="top" align="left">48.3&#x0002B;89&#x0002B;89 &#x0003D; 226.3</td>
<td valign="top" align="left">29.4&#x0002B;29.4&#x0002B;54.7&#x0002B;31&#x0002B;24 &#x0003D; 169.5</td>
<td valign="top" align="left">95</td>
<td valign="top" align="left">&#x02014;</td>
<td valign="top" align="left">28</td>
</tr>
<tr>
<td valign="top" align="left">Input image resolution</td>
<td valign="top" align="left">256 &#x000D7; 256, 256 &#x000D7; 256, 352 &#x000D7; 352</td>
<td valign="top" align="left">448 &#x000D7; 448</td>
<td valign="top" align="left">448 &#x000D7; 448</td>
<td valign="top" align="left">&#x02014;</td>
<td valign="top" align="left">448 &#x000D7; 448</td>
</tr>
<tr>
<td valign="top" align="left">Loss function</td>
<td valign="top" align="left">Triplet, AM-softmax, LDAM</td>
<td valign="top" align="left">SoftTriple, Cross-entropy, BM-Softmax.</td>
<td valign="top" align="left">Cross-entropy</td>
<td valign="top" align="left">&#x02014;</td>
<td valign="top" align="left">Cross-entropy</td>
</tr>
<tr>
<td valign="top" align="left">F<sub>1</sub></td>
<td valign="top" align="left">0.757</td>
<td valign="top" align="left">0.735</td>
<td valign="top" align="left">0.689</td>
<td valign="top" align="left">0.687</td>
<td valign="top" align="left">0.682</td>
</tr>
<tr>
<td valign="top" align="left">Acc</td>
<td valign="top" align="left">0.845</td>
<td valign="top" align="left">0.837</td>
<td valign="top" align="left">0.793</td>
<td valign="top" align="left">0.799</td>
<td valign="top" align="left">0.786</td>
</tr>
<tr>
<td valign="top" align="left">Mca</td>
<td valign="top" align="left">0.787</td>
<td valign="top" align="left">0.761</td>
<td valign="top" align="left">0.706</td>
<td valign="top" align="left">0.725</td>
<td valign="top" align="left">0.693</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Note that the fourth place team did not respond to the post&#x02013;competition survey. We define as a feature extractor the part of the model that extracts the feature maps on which the classification is based. The feature extractor parameters are taken from the publications associated with the respective model architectures</italic>.</p>
</table-wrap-foot>
</table-wrap>
<p>Collectively, the top&#x02013;five teams used seven different base neural network architectures:</p>
<list list-type="bullet">
<list-item><p><bold>ResNeXt</bold> The ResNeXt architecture introduced by Xie et al. (<xref ref-type="bibr" rid="B60">2017</xref>) is a popular network architecture that extends ResNet (He et al., <xref ref-type="bibr" rid="B17">2016</xref>). It leverages the split&#x02013;transform&#x02013;merge (proposed in Inception; Szegedy et al., <xref ref-type="bibr" rid="B47">2015</xref>) to split the input into multiple blocks and then merge those blocks after convolution.</p></list-item>
<list-item><p><bold>ResNeXt-IBN-a</bold> Pan et al. (<xref ref-type="bibr" rid="B38">2018</xref>) proposed IBN-Net as an extension to any existing network&#x02014;in this case the ResNeXt architecture. IBN stands for Instance and Batch Normalization&#x02014;the main modifications used in IBN-Net to achieve domain/appearance invariance. This modificaiton is a simple way to increase both modeling and generalization capacity without increasing computational burden.</p></list-item>
<list-item><p><bold>SE-ResNeXt</bold> The SE network introduced by Hu et al. (<xref ref-type="bibr" rid="B24">2018</xref>) focuses on channel relationships instead of the spatial component of convolutional blocks. This is done by using the &#x0201C;Squeeze&#x02013;and&#x02013;Excitation&#x0201D; (SE) block, that adaptively recalibrates channel&#x02013;wise feature responses by explicitly modeling interdependencies among channels. In this case the standard convolutional blocks in the ResNeXt architecture are replaced by these new SE blocks.</p></list-item>
<list-item><p><bold>ResNeSt</bold> The ResNeSt architecture proposed by Zhang et al. (<xref ref-type="bibr" rid="B62">2020</xref>) is a variant of the ResNet model which instead stacks Split&#x02013;Attention blocks which are effectively channel&#x02013;wise attention on different network branches.</p></list-item>
<list-item><p><bold>TResNet</bold> The TResNet architecture proposed by Ridnik et al. (<xref ref-type="bibr" rid="B43">2020</xref>) is designed to be highly efficient in training time and inference time while achieving a better performance than a comparable ResNet.</p></list-item>
<list-item><p><bold>ECA-NFNet-L0</bold> The ECA-NFNet is a variant of the Normalization&#x02013;Free neural Network (NFNet; Brock et al., <xref ref-type="bibr" rid="B4">2021</xref>) with Efficient Channel Attention (ECA) layers (Wang et al., <xref ref-type="bibr" rid="B55">2020</xref>) instead of SE blocks, which results in one third of the number of parameters of the original NFNet.</p></list-item>
<list-item><p><bold>GENet</bold> The GENet proposed by Lin et al. (<xref ref-type="bibr" rid="B32">2020</xref>) is designed to be efficient when trained on a GPU. In fact, it achieves a similar performance, but is up to 6.4 times faster than EfficientNet (Tan and Le, <xref ref-type="bibr" rid="B49">2019</xref>).</p></list-item>
</list>
<p>Interestingly, the top&#x02013;two teams leveraged recently proposed deep metric learning losses in addition to the standard cross&#x02013;entropy loss used for classification. The goal of deep metric learning is to learn an embedding where the features extracted from examples of the same class (in this case, the same taxon) are closer than the ones extracted from examples of different classes. The issue with standard cross&#x02013;entropy loss preceded by a softmax is that it learns separable features that are not discriminative enough&#x02014;this problem is exacerbated in the Herbarium 2021 dataset where the training set is extremely long&#x02013;tailed and performance is measured on a relatively well-balanced test set. One way to produce a deep metric learning embedding is to cast it as an optimization problem with triplet constraints, which correspond to the Triplet loss: learning is performed on a set of three images, the anchor (the baseline image), the positive image (another image belonging to the same class as the anchor), and the negative image (an image belonging to a different class). The goal is then to have features which correspond to the anchor and the positive image (or images) close in the embedding space while the anchor and the negative image (or images) are far in the embedding space. However, this procedure is time consuming and it is very sensitive to the selection of anchor, positive, and negative images. As a result there has been a number of loss functions proposed as extensions of the standard cross&#x02013;entropy loss, that achieve the objective of the distance metric learning paradigm without having to compare multiple image samples in embedding space: Additive Margin Softmax loss (AM&#x02013;softmax; Wang et al., <xref ref-type="bibr" rid="B54">2018</xref>), Balanced Meta&#x02013;Softmax loss (BM&#x02013;softmax; Ren et al., <xref ref-type="bibr" rid="B42">2020</xref>), and SoftTriple loss (Qian et al., <xref ref-type="bibr" rid="B41">2019</xref>) are examples. Finally the Label&#x02013;Distribution&#x02013;Aware Margin Loss (LDAM; Cao et al., <xref ref-type="bibr" rid="B6">2019</xref>) is designed to replace the cross&#x02013;entropy loss&#x02014;it is designed specifically for the case in which the training dataset is heavily imbalanced while the testing criterion requires good generalization on less frequent classes.</p>
<p>Regarding the losses, unfortunately none of the teams leveraged the provided hierarchical labels. In <xref ref-type="table" rid="T3">Table 3</xref>, we highlight the potential increase in performance that could be achieved by using them. In fact, there is clearly a substantial improvement when comparing the performance of the baseline model trained with a standard cross&#x02013;entropy loss to the performance achieved when training the same model with the marginalization loss (Section 2.2). The marginalization loss is trivial to extend to any of the loss functions used by the competitors other than cross&#x02013;entropy loss (<xref ref-type="table" rid="T2">Table 2</xref>).</p>
<table-wrap position="float" id="T3">
<label>Table 3</label>
<caption><p>Ablation study for marginalization loss utilizing hierarchical label information.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th valign="top" align="left"><bold>Model</bold></th>
<th valign="top" align="center"><bold>F<sub><bold>1</bold></sub></bold></th>
<th valign="top" align="center"><bold>Acc</bold></th>
<th valign="top" align="center"><bold>Mca</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Baseline</td>
<td valign="top" align="center">0.442</td>
<td valign="top" align="center">0.543</td>
<td valign="top" align="center">0.485</td>
</tr>
<tr>
<td valign="top" align="left">Baseline with marginalization loss</td>
<td valign="top" align="center">0.494</td>
<td valign="top" align="center">0.599</td>
<td valign="top" align="center">0.534</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec>
<title>3.2. Performance on Difficult Examples</title>
<p>The top&#x02013;five competition models accurately predicted the correct taxa for the examples presented in Section 2.1.1: <italic>Abarema brahcystahya</italic> (<xref ref-type="fig" rid="F3">Figure 3</xref>), used to illustrate different reproductive stages, had an average top-1 accuracy of 0.914 (test images <italic>n</italic> = 10; training images <italic>n</italic> = 33); <italic>Arbutus xalapensis</italic>, used to illustrate variation in specimen preparation (<xref ref-type="fig" rid="F4">Figure 4</xref>) had an average top-1 accuracy of 0.94 (test images <italic>n</italic> = 7; training images <italic>n</italic> = 297); and the <italic>Alyssum</italic> species, used to illustrate similar morphology among closely related taxa (<xref ref-type="fig" rid="F5">Figure 5</xref>), had an average top-1 accuracy of 0.926 (test images <italic>n</italic> = 38; average training images per species <italic>n</italic> = 40.22, range = 2&#x02013;154).</p>
</sec>
<sec>
<title>3.3. Patristic Classification Error</title>
<p>The magnitude of classification error can be measured by the patristic distance between the expected and predicted taxa. When the predictions of the top&#x02013;five models are incorrect, the wrongly predicted taxon is usually one that is phylogenetically close to the expected taxon (i.e., low patristic distance between predicted and expected taxa). For instance, if all model predictions within a maximum patristic distance of 10 million years (My) from the expected taxon are considered correct, then all five top models display similar accuracy (0.77&#x02013;0.86; <xref ref-type="fig" rid="F8">Figure 8</xref>). On the other hand, when the threshold is 30 My, which is close to the median patristic distance between sister genera (31.628 My), the error rate is less than 10% for the top&#x02013;two models (<xref ref-type="fig" rid="F8">Figure 8</xref>). Thus, the models are generally correct at the genus&#x02013;level and more than half of the original error is due to incorrect classification of taxa within genera.</p>
<fig id="F8" position="float">
<label>Figure 8</label>
<caption><p>Model performance measured by different phylogenetic proximity thresholds. Top-1 error is calculated by counting all predictions that are within the patristic distance threshold as successes. The vertical dashed lines represent top-1 error at 0.01, 0.05, and 0.10, respectively.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0008.tif"/>
</fig>
<p>When evaluated in the light of patristic classification error, the third place model does not appear to behave like the other top&#x02013;five models (<xref ref-type="fig" rid="F8">Figure 8</xref>): perhaps the features it extracts are less correlated with phylogeny than the features extracted by the other top models. Given that the fifth place model uses the same SE-ResNeXt base architecture and cross&#x02013;entropy loss function, the deviant performance could, perhaps, be attributed to training parameters.</p>
<p>Examination of the erroneous predictions made by the top performing model do not indicate any phylogenetic clustering of errors&#x02014;demonstrating that the top model performs equally well (or equally poorly) on all types of plants in the dataset (<xref ref-type="fig" rid="F9">Figure 9</xref>). If a botanist was to be provided with the low&#x02013;resolution input images used by the top model, they would be unlikely to perform as uniformly as the model: taxa in some orders are almost exclusively differentiated by features occupying only a fraction of a pixel at that resolution (e.g., Poales) while taxa in other orders are more easily differentiated at that resolution (e.g., Rosales).</p>
<fig id="F9" position="float">
<label>Figure 9</label>
<caption><p>The relationship among dataset properties and incorrect model predictions for the top performing model. The phylogenetic relationship among the expected taxa (x&#x02013;axis) is represented by the right ladderized phylogenetic tree for all genera in the dataset (Jin and Qian, <xref ref-type="bibr" rid="B26">2019</xref>). The y&#x02013;axis indicates the identification error&#x02014;expressed as patristic distance between the expected and predicted classifications. The number of training images for each expected taxon is indicated by marker color and visualized as a histogram in the right panel. The median patristic distance between sister genera is represented by a solid horizontal gray line with gray boxes indicating the 10&#x02013;90, 20&#x02013;80, 30&#x02013;70, and 40&#x02013;60 decile ranges. The top ten angiosperm, top gymnosperm, top fern, and top lycophyte orders, as measured by the number of training images, are labeled. Results for 628 of 243,020 (0.26%) test images are not displayed because those taxa could not be located in the reference phylogenetic tree (Jin and Qian, <xref ref-type="bibr" rid="B26">2019</xref>).</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0009.tif"/>
</fig>
<p>Prediction errors less than the median patristic distance between sister genera (31.628 My; solid horizontal gray line in <xref ref-type="fig" rid="F9">Figure 9</xref>), are the sorts of errors that botanists commonly make (i.e., misidentifying taxa within genera). Some of these model errors may be the result of uncaught labeling errors in our dataset. Prediction errors above the 90<italic>th</italic> decile of the patristic distance between sister genera (112.160 My; outer gray box in <xref ref-type="fig" rid="F9">Figure 9</xref>) are errors that botanists rarely make and, thus, are unlikely to be attributable to incorrect dataset labeling.</p>
</sec>
<sec>
<title>3.4. Factors Contributing to Prediction Error</title>
<p>For the top performing model, the number of expected taxon training images appears to be associated with prediction failure, but the relationship is not absolute: there are cases, particularly common in the Polypodiales, in which the number of training images is high (dark blue circles in <xref ref-type="fig" rid="F9">Figure 9</xref>) and the patristic classification error is high. The relationship between model accuracy and number of training images is more straightforward (<xref ref-type="fig" rid="F10">Figure 10</xref>): the top performing model shows poor accuracy for taxa with only two training images (accuracy = 56.0%, <italic>n</italic> = 7,745), while the accuracy substantially increases with more training images (e.g., accuracy = 79.8% for taxa with eight training images, <italic>n</italic> = 2,147). The top-1 accuracy of the second and fifth place models is less than 50% (42.7&#x02013;46.7%, <italic>n</italic> = 7,745) with two training images, while a similar boost in accuracy with more training data is observed (66.3&#x02013;77.1% with eight training images, <italic>n</italic> = 2,147). Model accuracy increases with the number of training images to different degrees across the top&#x02013;five models. The top&#x02013;two model shows a consistent boost in its performance as training images increases (<italic>n</italic> = 2&#x02013;3309), whereas other models display inconsistent performance boosts after <italic>n</italic> = 100 training images&#x02014;the top&#x02013;four model shows a consistent decrease in performance after <italic>n</italic> = 100 training images.</p>
<fig id="F10" position="float">
<label>Figure 10</label>
<caption><p>Mean performance of the top&#x02013;five models by number of training images. Taxa are aggregated based on their corresponding number of training images. The test time performance is then visualized as the mean of the top-1-accuracy for all taxa in a specific bin, the error bars correspond to the average standard error for each bin.</p></caption>
<graphic mimetype="image" mime-subtype="tiff" xlink:href="fpls-12-787127-g0010.tif"/>
</fig>
<p>Another factor affecting model predication accuracy is specimen quality: we examined the 144 specimen images in the test dataset that produced egregiously incorrect (i.e., patristic classification error greater than 600 My) predictions from the top performing model, and compared them to an equally sized randomly&#x02013;sampled set of images with correct predictions. About 9.1% of the egregiously misclassified images were not good plant specimens: 0.7% were photographs of living plants, 0.7% were botanical illustrations, 3.5% lack plant materials entirely, and 4.2% were closed packets that obscured all plant materials from view. None of the correctly predicted specimen images had the above issues. Among the egregiously misclassified specimen images with visible plant materials (90.9%), nearly half (46.6%) consist entirely of plant fragments (e.g., single organs like fruits, buds, leaves, inflorescence, etc.) and a quarter (25.7%) are diminutive plant taxa&#x02014;that remain small at full maturity&#x02014;and therefor cover only a tiny fraction of specimen image. In contrast, 3.6% of correctly predicted specimen images consist entirely of plant fragments and none were diminutive plants.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>4. Discussion</title>
<sec>
<title>4.1. Competition Results</title>
<p>The Herbarium 2021 Half&#x02013;Earth challenge is the richest plant dataset in the world for fine&#x02013;grained visual categorization, but it pushes the limits of contemporary machine learning&#x02014;automated herbarium specimen image classification is a challenge that involves differentiating among plant taxa with subtle differences in color, texture, and shape. When compared to other fine&#x02013;grained image datasets such as ImageNet (distinct classes easily classified by the general public; Deng et al., <xref ref-type="bibr" rid="B11">2009</xref>) or iNaturalist (distinct classes that are easier to classify due to their spread across different kingdoms of life; Horn et al., <xref ref-type="bibr" rid="B23">2021</xref>), the difficulty of classifying herbarium specimen images is apparent. The high number (64,500) and imbalanced distribution (imbalance factor = 1,654.5) of classes in the Herbarium 2021 dataset, makes this task especially challenging given the numerous classes with few images&#x02014;nearly half (49.1%) of the taxa have less than 10 training images. Despite these challenges, the deep learning models submitted to the competition demonstrated performance far beyond our expectations: macro F<sub>1</sub>-score = 0.76 and top-1-error = 15.5%.</p>
<p>Most recently ViT-G/14 (Dosovitskiy et al., <xref ref-type="bibr" rid="B13">2021</xref>) achieved a top-1-error of 9.55% on ImageNet&#x02014;the most widely used image classification dataset. Considering that our dataset is much more unevenly distributed and has 60 times more classes than ImageNet,the top-1-error of 15.5% for the Herbarium 2021 Half&#x02013;Earth challenge is quite remarkable (<xref ref-type="table" rid="T2">Table 2</xref>). For taxa with more than 50 training images (<italic>n</italic> = 10,355 taxa), the top-1-error (10.4%) of the top performance model is comparable to the state-of-the-art top-1-error of ImageNet (9.55%)&#x02014;even with 10 times more classes and 7.8 times fewer parameters than ViT-G/14 (230 M vs. 1,800 M). The iNaturalist 2021 (Horn et al., <xref ref-type="bibr" rid="B23">2021</xref>) fine&#x02013;grained visual categorization dataset is similar to Herbarium 2021 in many ways, but it includes only ten thousand taxa with a more balanced training data distribution (&#x0003E;100 training images per taxon) and incorporates image geo&#x02013;locations. In contrast, Herbarium 2021 does not include collection locations. In the Kaggle competition, the best model for iNaturalist 2021 had a top-1-error rate of 4.4%. If the Herbarium 2021 dataset had a more balanced distribution of training images, we see it having the potential to become another rich source of fine&#x02013;grained visual categorization tests.</p>
<p>Finally we would like to point out that the competition was particularly difficult for competitors who did not have the computational resources to train large models on this amount of data&#x02014;training a large model on this dataset is quite time consuming: for example training the baseline model took around 120h on an NVIDIA Titan X GPU. As can be seen in <xref ref-type="table" rid="T2">Table 2</xref>, the top-5 competitors&#x00027; performance seems to be correlated with the number of parameters of the feature extractor.</p>
</sec>
<sec>
<title>4.2. Future Directions</title>
<p>There a multiple future directions that can be explored within the scope of fine&#x02013;grained herbarium classification:</p>
<list list-type="bullet">
<list-item><p>Automated analysis of digitized natural history collections may help reduce the bottlenecks in identifying new species held in collections: herbaria are thought to already house specimens of half of the plant species that have not yet been formally described in the scientific literature (Bebber et al., <xref ref-type="bibr" rid="B2">2010</xref>). There is an incredible backlog in specimen identification and curation in herbaria and many lack staff and taxonomic expertise to readily identify all of their specimens. With this urgent need in mind, we believe that there is an opportunity to facilitate the work of botanical experts to enable them to focus on the most critical tasks that cannot be automated. One useful approach may be to build a dataset that includes unlabeled data so that competitors could explore approaches related to semi&#x02013;supervised learning or active learning rather than limiting competitions to straightforward supervised learning tasks. Furthermore, systems to accurately estimate well-calibrated uncertainties linked to the taxon prediction task would be extremely useful to make sure that we prioritize specimens most needing attention from expert botanists.</p></list-item>
<list-item><p>It may be possible to leverage the digitized data stored in the herbaria to classify pictures of living plants. Overcoming the distribution shift between training on herbarium sheet images and testing on images of live plants is non-trivial, nevertheless recent advancements in generative models and domain adaptation can be effectively applied to such a scenario.</p></list-item>
<list-item><p>Future Kaggle challenges should encourage engagement between different research communities, such as computer vision scientists and botanists. Computer vision scientists often adopt an approach aimed at maximizing algorithm performance in terms of the evaluation metrics, but they may be unaware of domain specific knowledge, such as the patristic distance, that can be used to both improve model interpretability and performance. On the other hand, botanists may not be aware of the latest advances in computer science that may boost model performance.</p></list-item>
<list-item><p>Although large datasets increase the difficulty of the competition and push the boundaries of automatic taxon recognition, they exclude participants without access to a large computational resources for training machine learning models. As a result, future Kaggle challenges could be designed so that they can be split in multiple parts with at least some of the parts computationally accessible to all (e.g., a dataset of selected families or orders, or a dataset with a cap on the maximum number of images per taxon).</p></list-item>
</list>
</sec>
</sec>
<sec sec-type="conclusions" id="s5">
<title>5. Conclusion</title>
<p>We have created the Herbarium 2021 Half&#x02013;Earth dataset to enable the development of better automatic taxon recognition models. The development of models to automatically identify specimens will reduce the species identification bottleneck and has the potential to improve both the quality and accelerate the pace of biodiversity research.</p>
<p>In the future, we would like to expand the dataset to include specimens collected world&#x02013;wide. There are more than 35 million digitized specimens in electronic databases representing more than 80% of the known vascular plant diversity.</p>
</sec>
<sec sec-type="data-availability" id="s6">
<title>Data Availability Statement</title>
<p>The datasets presented in this study can be found in online repositories. The names of the repository/repositories and accession number(s) can be found in the article/supplementary material.</p>
</sec>
<sec id="s7">
<title>Author Contributions</title>
<p>RL, DL, JP, KW and SD&#x00027;A wrote the manuscript in consultation with JDW, SB, FM, and BA. RL and DL prepared the dataset and the competition in consultation with JDW, SB, FM, and BA. KW, JJW, MT, RP, TG, GB, GG, AF, DR, and YB selected and provided the data from their respective institutions. SB and BA conceived the original idea. All authors contributed to the article and approved the submitted version.</p>
</sec>
<sec sec-type="funding-information" id="s8">
<title>Funding</title>
<p>This work was partially funded by National Science Foundation (USA) grant DEB 2054684.</p>
</sec>
<sec sec-type="COI-statement" id="conf1">
<title>Conflict of Interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest. The handling editor declared a shared research group [FGVC virtual lab] with one of the authors [SB] at time of review.</p>
</sec>
<sec sec-type="disclaimer" id="s9">
<title>Publisher&#x00027;s Note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
</body>
<back>
<ack><p>We would like to thank Kiat Chuan Tan from Google and the team at Kaggle (Walter Reade and Maggie Demkin) for their generous support in making this challenge possible. We would also like to thank everyone who entered the Herbarium 2021 Half&#x02013;Earth Challenge. We are particularly grateful to the teams that provided detailed information on the model architectures and training strategies behind their winning submissions.</p></ack>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Albani Rocchetti</surname> <given-names>G.</given-names></name> <name><surname>Armstrong</surname> <given-names>C. G.</given-names></name> <name><surname>Abeli</surname> <given-names>T.</given-names></name> <name><surname>Orsenigo</surname> <given-names>S.</given-names></name> <name><surname>Jasper</surname> <given-names>C.</given-names></name> <name><surname>Joly</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>Reversing extinction trends: new uses of (old) herbarium specimens to accelerate conservation action on threatened species</article-title>. <source>N. Phytol</source>. <volume>230</volume>, <fpage>433</fpage>&#x02013;<lpage>450</lpage>. <pub-id pub-id-type="doi">10.1111/nph.17133</pub-id><pub-id pub-id-type="pmid">33280123</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bebber</surname> <given-names>D. P.</given-names></name> <name><surname>Carine</surname> <given-names>M. A.</given-names></name> <name><surname>Wood</surname> <given-names>J. R. I.</given-names></name> <name><surname>Wortley</surname> <given-names>A. H.</given-names></name> <name><surname>Harris</surname> <given-names>D. J.</given-names></name> <name><surname>Prance</surname> <given-names>G. T.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Herbaria are a major frontier for species discovery</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A</source>. <volume>107</volume>, <fpage>22169</fpage>&#x02013;<lpage>22171</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.10118859441108</pub-id><pub-id pub-id-type="pmid">21135225</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brenskelle</surname> <given-names>L.</given-names></name> <name><surname>Guralnick</surname> <given-names>R. P.</given-names></name> <name><surname>Denslow</surname> <given-names>M.</given-names></name> <name><surname>Stucky</surname> <given-names>B. J.</given-names></name></person-group> (<year>2020</year>). <article-title>Maximizing human effort for analyzing scientific images: a case study using digitized herbarium sheets</article-title>. <source>Appl. Plant Sci</source>. <volume>8</volume>, <fpage>e11370</fpage>. <pub-id pub-id-type="doi">10.1002/aps3.11370</pub-id><pub-id pub-id-type="pmid">32626612</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Brock</surname> <given-names>A.</given-names></name> <name><surname>De</surname> <given-names>S.</given-names></name> <name><surname>Smith</surname> <given-names>S. L.</given-names></name> <name><surname>Simonyan</surname> <given-names>K.</given-names></name></person-group> (<year>2021</year>). <article-title>High-performance large-scale image recognition without normalization</article-title>, in <source>Proceedings of the International Conference on Machine Learning</source>.</citation>
</ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Calinger</surname> <given-names>K. M.</given-names></name> <name><surname>Queenborough</surname> <given-names>S.</given-names></name> <name><surname>Curtis</surname> <given-names>P. S.</given-names></name></person-group> (<year>2013</year>). <article-title>Herbarium specimens reveal the footprint of climate change on flowering trends across north-central North America</article-title>. <source>Ecol. Lett</source>. <volume>16</volume>, <fpage>1037</fpage>&#x02013;<lpage>1044</lpage>. <pub-id pub-id-type="doi">10.1111/ele.12135</pub-id><pub-id pub-id-type="pmid">23786499</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Cao</surname> <given-names>K.</given-names></name> <name><surname>Wei</surname> <given-names>C.</given-names></name> <name><surname>Gaidon</surname> <given-names>A.</given-names></name> <name><surname>Arechiga</surname> <given-names>N.</given-names></name> <name><surname>Ma</surname> <given-names>T.</given-names></name></person-group> (<year>2019</year>). <article-title>Learning imbalanced datasets with label-distribution-aware margin loss</article-title>, in <source>Proceedings of the Conference on Neural Information Processing Systems</source> (<publisher-loc>Vancouver, BC</publisher-loc>).</citation>
</ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carranza-Rojas</surname> <given-names>J.</given-names></name> <name><surname>Goeau</surname> <given-names>H.</given-names></name> <name><surname>Bonnet</surname> <given-names>P.</given-names></name> <name><surname>Mata-Montero</surname> <given-names>E.</given-names></name> <name><surname>Joly</surname> <given-names>A.</given-names></name></person-group> (<year>2017</year>). <article-title>Going deeper in the automated identification of herbarium specimens</article-title>. <source>BMC Evol. Biol</source>. <volume>17</volume>, <fpage>181</fpage>. <pub-id pub-id-type="doi">10.1186/s12862-017-1014-z</pub-id><pub-id pub-id-type="pmid">28797242</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Clark</surname> <given-names>J. Y.</given-names></name> <name><surname>Corney</surname> <given-names>D. P. A.</given-names></name> <name><surname>Tang</surname> <given-names>H. L.</given-names></name></person-group> (<year>2012</year>). <article-title>Automated plant identification using artificial neural networks</article-title>, in <source>2012 IEEE Symposium on Computational Intelligence in Bioinformatics and Computational Biology (CIBCB)</source> (<publisher-loc>San Diego, CA</publisher-loc>), <fpage>343</fpage>&#x02013;<lpage>348</lpage>. <pub-id pub-id-type="doi">10.1109/CIBCB.2012.6217250</pub-id><pub-id pub-id-type="pmid">14642663</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cope</surname> <given-names>J. S.</given-names></name> <name><surname>Corney</surname> <given-names>D.</given-names></name> <name><surname>Clark</surname> <given-names>J. Y.</given-names></name> <name><surname>Remagnino</surname> <given-names>P.</given-names></name> <name><surname>Wilkin</surname> <given-names>P.</given-names></name></person-group> (<year>2012</year>). <article-title>Plant species identification using digital morphometrics: a review</article-title>. <source>Expert Syst. Appl</source>. <volume>39</volume>, <fpage>7562</fpage>&#x02013;<lpage>7573</lpage>. <pub-id pub-id-type="doi">10.1016/j.eswa.2012.01.073</pub-id></citation>
</ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Davis</surname> <given-names>C. C.</given-names></name> <name><surname>Champ</surname> <given-names>J.</given-names></name> <name><surname>Park</surname> <given-names>D. S.</given-names></name> <name><surname>Breckheimer</surname> <given-names>I.</given-names></name> <name><surname>Lyra</surname> <given-names>G. M.</given-names></name> <name><surname>Xie</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>A new method for counting reproductive structures in digitized herbarium specimens using mask R-CNN</article-title>. <source>Front. Plant Sci</source>. <volume>11</volume>, <fpage>1129</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2020.01129</pub-id><pub-id pub-id-type="pmid">32849691</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Deng</surname> <given-names>J.</given-names></name> <name><surname>Dong</surname> <given-names>W.</given-names></name> <name><surname>Socher</surname> <given-names>R.</given-names></name> <name><surname>Li</surname> <given-names>L.-J.</given-names></name> <name><surname>Li</surname> <given-names>K.</given-names></name> <name><surname>Fei-Fei</surname> <given-names>L.</given-names></name></person-group> (<year>2009</year>). <article-title>ImageNet: a large-scale hierarchical image database</article-title>, in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Miami, FL</publisher-loc>), <fpage>248</fpage>&#x02013;<lpage>255</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2009.5206848</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Dillen</surname> <given-names>M.</given-names></name> <name><surname>Groom</surname> <given-names>Q.</given-names></name> <name><surname>Chagnoux</surname> <given-names>S.</given-names></name> <name><surname>G&#x000FC;ntsch</surname> <given-names>A.</given-names></name> <name><surname>Hardisty</surname> <given-names>A.</given-names></name> <name><surname>Haston</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>A benchmark dataset of herbarium specimen images with label data</article-title>. <source>Biodivers. Data J</source>. <volume>7</volume>, <fpage>e31817</fpage>. <pub-id pub-id-type="doi">10.3897/BDJ.7.e31817</pub-id><pub-id pub-id-type="pmid">30833825</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Dosovitskiy</surname> <given-names>A.</given-names></name> <name><surname>Beyer</surname> <given-names>L.</given-names></name> <name><surname>Kolesnikov</surname> <given-names>A.</given-names></name> <name><surname>Weissenborn</surname> <given-names>D.</given-names></name> <name><surname>Zhai</surname> <given-names>X.</given-names></name> <name><surname>Unterthiner</surname> <given-names>T.</given-names></name> <etal/></person-group>. (<year>2021</year>). <article-title>An image is worth 16x16 words: Transformers for image recognition at scale</article-title>, in <source>Proceedings of the International Conference on Learning Representations</source>.</citation>
</ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Freiberg</surname> <given-names>M.</given-names></name> <name><surname>Winter</surname> <given-names>M.</given-names></name> <name><surname>Gentile</surname> <given-names>A.</given-names></name> <name><surname>Zizka</surname> <given-names>A.</given-names></name> <name><surname>Muellner-Riehl</surname> <given-names>A. N.</given-names></name> <name><surname>Weigelt</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>LCVP, the leipzig catalogue of vascular plants, a new taxonomic reference list for all known vascular plants</article-title>. <source>Sci. Data</source> <volume>7</volume>, <fpage>416</fpage>. <pub-id pub-id-type="doi">10.1038/s41597-020-00702-z</pub-id><pub-id pub-id-type="pmid">33243996</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Funk</surname> <given-names>V. A.</given-names></name></person-group> (<year>2003</year>). <article-title>The importance of herbaria</article-title>. <source>Plant Sci. Bull</source>. <volume>49</volume>, <fpage>94</fpage>&#x02013;<lpage>95</lpage>.</citation>
</ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Go&#x000EB;au</surname> <given-names>H.</given-names></name> <name><surname>Mora-Fallas</surname> <given-names>A.</given-names></name> <name><surname>Champ</surname> <given-names>J.</given-names></name> <name><surname>Love</surname> <given-names>N. L. R.</given-names></name> <name><surname>Mazer</surname> <given-names>S. J.</given-names></name> <name><surname>Mata-Montero</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>A new fine-grained method for automated visual analysis of herbarium specimens: a case study for phenological data extraction</article-title>. <source>Appl. Plant Sci</source>. <volume>8</volume>, <fpage>e11368</fpage>. <pub-id pub-id-type="doi">10.1002/aps3.11368</pub-id><pub-id pub-id-type="pmid">32626610</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>He</surname> <given-names>K.</given-names></name> <name><surname>Zhang</surname> <given-names>X.</given-names></name> <name><surname>Ren</surname> <given-names>S.</given-names></name> <name><surname>Sun</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>Deep residual learning for image recognition</article-title>, in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Las Vegas, CA</publisher-loc>), <fpage>770</fpage>&#x02013;<lpage>778</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2016.90</pub-id><pub-id pub-id-type="pmid">32166560</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heberling</surname> <given-names>J. M.</given-names></name> <name><surname>Burke</surname> <given-names>D. J.</given-names></name></person-group> (<year>2019</year>). <article-title>Utilizing herbarium specimens to quantify historical mycorrhizal communities</article-title>. <source>Appl. Plant Sci</source>. <volume>7</volume>, <fpage>e01223</fpage>. <pub-id pub-id-type="doi">10.1002/aps3.1223</pub-id><pub-id pub-id-type="pmid">31024779</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heberling</surname> <given-names>J. M.</given-names></name> <name><surname>Miller</surname> <given-names>J. T.</given-names></name> <name><surname>Noesgaard</surname> <given-names>D.</given-names></name> <name><surname>Weingart</surname> <given-names>S. B.</given-names></name> <name><surname>Schigel</surname> <given-names>D.</given-names></name></person-group> (<year>2021</year>). <article-title>Data integration enables global biodiversity synthesis</article-title>. <source>Proc. Natl. Acad. Sci</source>. <volume>118</volume>, <fpage>e2018093118</fpage>. <pub-id pub-id-type="doi">10.1073/pnas.2018093118</pub-id><pub-id pub-id-type="pmid">33526679</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Heberling</surname> <given-names>J. M.</given-names></name> <name><surname>Prather</surname> <given-names>L. A.</given-names></name> <name><surname>Tonsor</surname> <given-names>S. J.</given-names></name></person-group> (<year>2019</year>). <article-title>The changing uses of herbarium data in an era of global change: an overview using automated content analysis</article-title>. <source>Bioscience</source> <volume>69</volume>, <fpage>812</fpage>&#x02013;<lpage>822</lpage>. <pub-id pub-id-type="doi">10.1093/biosci/biz094</pub-id></citation>
</ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hogeweg</surname> <given-names>L.</given-names></name> <name><surname>Schermer</surname> <given-names>M.</given-names></name> <name><surname>Pieterse</surname> <given-names>S.</given-names></name> <name><surname>Roeke</surname> <given-names>T.</given-names></name> <name><surname>Wilfred</surname> <given-names>G.</given-names></name></person-group> (<year>2019</year>). <article-title>Machine learning model for identifying Dutch/Belgian biodiversity</article-title>. <source>Biodivers. Inform. Sci. Standards</source> <volume>3</volume>, <fpage>e39229</fpage>. <pub-id pub-id-type="doi">10.3897/biss.3.39229</pub-id></citation>
</ref>
<ref id="B22">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Horn</surname> <given-names>G. V.</given-names></name> <name><surname>Aodha</surname> <given-names>O. M.</given-names></name> <name><surname>Song</surname> <given-names>Y.</given-names></name> <name><surname>Shepard</surname> <given-names>A.</given-names></name> <name><surname>Adam</surname> <given-names>H.</given-names></name> <name><surname>Perona</surname> <given-names>P.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>The inaturalist species classification and detection dataset</article-title>, in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Salt Lake City, UT</publisher-loc>: <publisher-name>IEEE</publisher-name>), <fpage>8769</fpage>&#x02013;<lpage>8778</lpage>. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00914</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Horn</surname> <given-names>G. V.</given-names></name> <name><surname>Cole</surname> <given-names>E.</given-names></name> <name><surname>Beery</surname> <given-names>S.</given-names></name> <name><surname>Wilber</surname> <given-names>K.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name> <name><surname>Aodha</surname> <given-names>O. M.</given-names></name></person-group> (<year>2021</year>). <article-title>Benchmarking representation learning for natural world image collections</article-title>, in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>. <pub-id pub-id-type="doi">10.1109/CVPR46437.2021.01269</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Shen</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>G.</given-names></name></person-group> (<year>2018</year>). <article-title>Squeeze-and-excitation networks</article-title>. <source>arXiv preprint arXiv:1709.01507</source>. <pub-id pub-id-type="doi">10.1109/CVPR.2018.00745</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="book"><person-group person-group-type="author"><collab>Intergovernmental Science-Policy Platform on Biodiversity and Ecosystem Services</collab></person-group> (<year>2019</year>). <source>Summary for Policymakers of the Global Assessment Report on Biodiversity and Ecosystem Services (Summary for Policy Makers)</source>.</citation>
</ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jin</surname> <given-names>Y.</given-names></name> <name><surname>Qian</surname> <given-names>H.</given-names></name></person-group> (<year>2019</year>). <article-title>V.PhyloMaker: an R package that can generate very large phylogenies for vascular plants</article-title>. <source>Ecography</source> <volume>42</volume>, <fpage>1353</fpage>&#x02013;<lpage>1359</lpage>. <pub-id pub-id-type="doi">10.1111/ecog.04434</pub-id></citation>
</ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Joly</surname> <given-names>A.</given-names></name> <name><surname>Bonnet</surname> <given-names>P.</given-names></name> <name><surname>Go&#x000EB;au</surname> <given-names>H.</given-names></name> <name><surname>Barbe</surname> <given-names>J.</given-names></name> <name><surname>Selmi</surname> <given-names>S.</given-names></name> <name><surname>Champ</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>A look inside the pl&#x00040;ntnet experience</article-title>. <source>Multimedia Syst</source>. <volume>22</volume>, <fpage>751</fpage>&#x02013;<lpage>766</lpage>. <pub-id pub-id-type="doi">10.1007/s00530-015-0462-9</pub-id></citation>
</ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Joppa</surname> <given-names>L. N.</given-names></name> <name><surname>Roberts</surname> <given-names>D. L.</given-names></name> <name><surname>Pimm</surname> <given-names>S. L.</given-names></name></person-group> (<year>2011</year>). <article-title>How many species of flowering plants are there?</article-title> <source>Proc. R. Soc. B Biol. Sci</source>. <volume>278</volume>, <fpage>554</fpage>&#x02013;<lpage>559</lpage>. <pub-id pub-id-type="doi">10.1098/rspb.2010.1004</pub-id><pub-id pub-id-type="pmid">20610425</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kho</surname> <given-names>S. J.</given-names></name> <name><surname>Manickam</surname> <given-names>S.</given-names></name> <name><surname>Malek</surname> <given-names>S.</given-names></name> <name><surname>Mosleh</surname> <given-names>M.</given-names></name> <name><surname>Dhillon</surname> <given-names>S. K.</given-names></name></person-group> (<year>2017</year>). <article-title>Automated plant identification using artificial neural network and support vector machine</article-title>. <source>Front. Life Sci</source>. <volume>10</volume>, <fpage>98</fpage>&#x02013;<lpage>107</lpage>. <pub-id pub-id-type="doi">10.1080/21553769.2017.1412361</pub-id></citation>
</ref>
<ref id="B30">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Kumar</surname> <given-names>S.</given-names></name> <name><surname>Zheng</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Hierarchical category detector for clothing recognition from visual data</article-title>, in <source>Proceedings, IEEE International Conference on Computer Vision Workshops</source> (<publisher-loc>Venice</publisher-loc>), <fpage>2306</fpage>&#x02013;<lpage>2312</lpage>. <pub-id pub-id-type="doi">10.1109/ICCVW.2017.272</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lang</surname> <given-names>P. L. M.</given-names></name> <name><surname>Willems</surname> <given-names>F. M.</given-names></name> <name><surname>Scheepens</surname> <given-names>J. F.</given-names></name> <name><surname>Burbano</surname> <given-names>H. A.</given-names></name> <name><surname>Bossdorf</surname> <given-names>O.</given-names></name></person-group> (<year>2019</year>). <article-title>Using herbaria to study global environmental change</article-title>. <source>N. Phytol</source>. <volume>221</volume>, <fpage>110</fpage>&#x02013;<lpage>122</lpage>. <pub-id pub-id-type="doi">10.1111/nph.15401</pub-id><pub-id pub-id-type="pmid">30160314</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>M.</given-names></name> <name><surname>Chen</surname> <given-names>H.</given-names></name> <name><surname>Sun</surname> <given-names>X.</given-names></name> <name><surname>Qian</surname> <given-names>Q.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Jin</surname> <given-names>R.</given-names></name></person-group> (<year>2020</year>). <article-title>Neural architecture design for GPU-efficient networks</article-title>. <source>arXiv preprint arXiv:2006.14090</source>.</citation>
</ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Little</surname> <given-names>D. P.</given-names></name> <name><surname>Tulig</surname> <given-names>M.</given-names></name> <name><surname>Tan</surname> <given-names>K. C.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name> <name><surname>Kaeser-Chen</surname> <given-names>C.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>An algorithm competition for automatic species identification from herbarium specimens</article-title>. <source>Appl. Plant Sci</source>. <volume>8</volume>, <fpage>e11365</fpage>. <pub-id pub-id-type="doi">10.1002/aps3.11365</pub-id><pub-id pub-id-type="pmid">32626608</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lorieul</surname> <given-names>T.</given-names></name> <name><surname>Pearson</surname> <given-names>K. D.</given-names></name> <name><surname>Ellwood</surname> <given-names>E. R.</given-names></name> <name><surname>Goeau</surname> <given-names>H.</given-names></name> <name><surname>Molino</surname> <given-names>J.-F.</given-names></name> <name><surname>Sweeney</surname> <given-names>P. W.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Toward a large-scale and deep phenological stage annotation of herbarium specimens: Case studies from temperate, tropical, and equatorial floras</article-title>. <source>Appl. Plant Sci</source>. <volume>7</volume>, <fpage>e01233</fpage>. <pub-id pub-id-type="doi">10.1002/aps3.1233</pub-id><pub-id pub-id-type="pmid">30937225</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nasir</surname> <given-names>A.</given-names></name> <name><surname>Rahman</surname> <given-names>M. A.</given-names></name> <name><surname>Mat</surname> <given-names>N.</given-names></name> <name><surname>Mamat</surname> <given-names>R.</given-names></name></person-group> (<year>2014</year>). <article-title>Automatic identification of <italic>Ficus deltoidea</italic> Jack (Moraceae) varieties based on leaf</article-title>. <source>Math. Models Methods Appl. Sci</source>. <volume>8</volume>, <fpage>121</fpage>. <pub-id pub-id-type="doi">10.5539/mas.v8n5p121</pub-id></citation>
</ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nelson</surname> <given-names>G.</given-names></name> <name><surname>Ellis</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>The history and impact of digitization and digital data mobilization on biodiversity research</article-title>. <source>Philos. Trans. R. Soc. B Biol. Sci</source>. <volume>374</volume>, <fpage>20170391</fpage>. <pub-id pub-id-type="doi">10.1098/rstb.2017.0391</pub-id><pub-id pub-id-type="pmid">30455209</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nic Lughadha</surname> <given-names>E.</given-names></name> <name><surname>Bachman</surname> <given-names>S. P.</given-names></name> <name><surname>Leao</surname> <given-names>T. C. C.</given-names></name> <name><surname>Forest</surname> <given-names>F.</given-names></name> <name><surname>Halley</surname> <given-names>J. M.</given-names></name> <name><surname>Moat</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Extinction risk and threats to plants and fungi</article-title>. <source>Plants People Planet</source> <volume>2</volume>, <fpage>389</fpage>&#x02013;<lpage>408</lpage>. <pub-id pub-id-type="doi">10.1002/ppp3.10146</pub-id><pub-id pub-id-type="pmid">19218582</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>X.</given-names></name> <name><surname>Luo</surname> <given-names>P.</given-names></name> <name><surname>Shi</surname> <given-names>J.</given-names></name> <name><surname>Tang</surname> <given-names>X.</given-names></name></person-group> (<year>2018</year>). <article-title>Two at once: enhancing learning and generalization capacities via IBN-Net</article-title>. <source>arXiv preprint arXiv:1807.09441</source>. <pub-id pub-id-type="doi">10.1007/978-3-030-01225-0_29</pub-id></citation>
</ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pearson</surname> <given-names>K. D.</given-names></name> <name><surname>Nelson</surname> <given-names>G.</given-names></name> <name><surname>Aronson</surname> <given-names>M. F. J.</given-names></name> <name><surname>Bonnet</surname> <given-names>P.</given-names></name> <name><surname>Brenskelle</surname> <given-names>L.</given-names></name> <name><surname>Davis</surname> <given-names>C. C.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Machine learning using digitized herbarium specimens to advance phenological research</article-title>. <source>Bioscience</source> <volume>70</volume>, <fpage>610</fpage>&#x02013;<lpage>620</lpage>. <pub-id pub-id-type="doi">10.1093/biosci/biaa044</pub-id><pub-id pub-id-type="pmid">32665738</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pryer</surname> <given-names>K. M.</given-names></name> <name><surname>Tomasi</surname> <given-names>C.</given-names></name> <name><surname>Wang</surname> <given-names>X.</given-names></name> <name><surname>Meineke</surname> <given-names>E. K.</given-names></name> <name><surname>Windham</surname> <given-names>M. D.</given-names></name></person-group> (<year>2020</year>). <article-title>Using computer vision on herbarium specimen images to discriminate among closely related horsetails (<italic>Equisetum</italic>)</article-title>. <source>Appl. Plant Sci</source>. <volume>8</volume>, <fpage>e11372</fpage>. <pub-id pub-id-type="doi">10.1002/aps3.11372</pub-id><pub-id pub-id-type="pmid">32626613</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Qian</surname> <given-names>Q.</given-names></name> <name><surname>Shang</surname> <given-names>L.</given-names></name> <name><surname>Sun</surname> <given-names>B.</given-names></name> <name><surname>Hu</surname> <given-names>J.</given-names></name> <name><surname>Li</surname> <given-names>H.</given-names></name> <name><surname>Jin</surname> <given-names>R.</given-names></name></person-group> (<year>2019</year>). <article-title>Softtriple loss: deep metric learning without triplet sampling</article-title>, in <source>Proceedings of the International Conference on Computer Vision</source> (<publisher-loc>Seoul</publisher-loc>). <pub-id pub-id-type="doi">10.1109/ICCV.2019.00655</pub-id><pub-id pub-id-type="pmid">33831595</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ren</surname> <given-names>J.</given-names></name> <name><surname>Yu</surname> <given-names>C.</given-names></name> <name><surname>Sheng</surname> <given-names>S.</given-names></name> <name><surname>Ma</surname> <given-names>X.</given-names></name> <name><surname>Zhao</surname> <given-names>H.</given-names></name> <name><surname>Yi</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Balanced meta-softmax for long-tailed visual recognition</article-title>, in <source>Proceedings of the Conference on Neural Information Processing Systems</source>.</citation>
</ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ridnik</surname> <given-names>T.</given-names></name> <name><surname>Lawen</surname> <given-names>H.</given-names></name> <name><surname>Noy</surname> <given-names>A.</given-names></name> <name><surname>Friedman</surname> <given-names>I.</given-names></name></person-group> (<year>2020</year>). <article-title>TResNet: high performance GPU-dedicated architecture</article-title>. <source>arXiv preprint arXiv:2003.13630</source>. <pub-id pub-id-type="doi">10.1109/WACV48630.2021.00144</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schuettpelz</surname> <given-names>E.</given-names></name> <name><surname>Frandsen</surname> <given-names>P. B.</given-names></name> <name><surname>Dikow</surname> <given-names>R. B.</given-names></name> <name><surname>Brown</surname> <given-names>A.</given-names></name> <name><surname>Orli</surname> <given-names>S.</given-names></name> <name><surname>Peters</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Applications of deep convolutional neural networks to digitized natural history collections</article-title>. <source>Biodivers. Data J</source>. <volume>5</volume>, <fpage>e21139</fpage>. <pub-id pub-id-type="doi">10.3897/BDJ.5.e21139</pub-id><pub-id pub-id-type="pmid">29200929</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="book"><person-group person-group-type="author"><collab>Secretariat of the Convention on Biological Diversity</collab></person-group> (<year>2007</year>). <source>Guide to the Global Taxonomy Initiative</source>.</citation>
</ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stefanaki</surname> <given-names>A.</given-names></name> <name><surname>Porck</surname> <given-names>H.</given-names></name> <name><surname>Grimaldi</surname> <given-names>I. M.</given-names></name> <name><surname>Thurn</surname> <given-names>N.</given-names></name> <name><surname>Pugliano</surname> <given-names>V.</given-names></name> <name><surname>Kardinaal</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2019</year>). <article-title>Breaking the silence of the 500-year-old smiling garden of everlasting flowers: the En Tibi book herbarium</article-title>. <source>PLoS ONE</source> <volume>14</volume>, <fpage>e0217779</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0217779</pub-id><pub-id pub-id-type="pmid">31242215</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Szegedy</surname> <given-names>C.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Jia</surname> <given-names>Y.</given-names></name> <name><surname>Sermanet</surname> <given-names>P.</given-names></name> <name><surname>Reed</surname> <given-names>S.</given-names></name> <name><surname>Anguelov</surname> <given-names>D.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Going deeper with convolutions</article-title>, in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source> (<publisher-loc>Boston, MA</publisher-loc>). <pub-id pub-id-type="doi">10.1109/CVPR.2015.7298594</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>K. C.</given-names></name> <name><surname>Liu</surname> <given-names>Y.</given-names></name> <name><surname>Ambrose</surname> <given-names>B.</given-names></name> <name><surname>Tulig</surname> <given-names>M.</given-names></name> <name><surname>Belongie</surname> <given-names>S.</given-names></name></person-group> (<year>2019</year>). <article-title>The herbarium challenge 2019 dataset</article-title>, in <source>CVPRW, 6th Fine-Grained Visual Categorization Workshop (FGVC6)</source> (<publisher-loc>Long Beach, CA</publisher-loc>).</citation>
</ref>
<ref id="B49">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Tan</surname> <given-names>M.</given-names></name> <name><surname>Le</surname> <given-names>Q. V.</given-names></name></person-group> (<year>2019</year>). <article-title>EfficientNet: rethinking model scaling for convolutional neural networks</article-title>, in <source>Proceedings of the International Conference on Machine Learning</source> (<publisher-loc>Long Beach, CA</publisher-loc>).</citation>
</ref>
<ref id="B50">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Thiers</surname> <given-names>B. M.</given-names></name></person-group> (<year>2021</year>). <source>The World&#x00027;s Herbaria 2020: A Summary Report Based on Data From Index Herbariorum</source>. Technical report, The New York Botanical Garden, New York, NY.</citation>
</ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ubbens</surname> <given-names>J. R.</given-names></name> <name><surname>Stavness</surname> <given-names>I.</given-names></name></person-group> (<year>2017</year>). <article-title>Deep plant phenomics: a deep learning platform for complex plant phenotyping tasks</article-title>. <source>Front. Plant Sci</source>. <volume>8</volume>, <fpage>1190</fpage>. <pub-id pub-id-type="doi">10.3389/fpls.2017.01190</pub-id><pub-id pub-id-type="pmid">29375612</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Unger</surname> <given-names>J.</given-names></name> <name><surname>Merhof</surname> <given-names>D.</given-names></name> <name><surname>Renner</surname> <given-names>S.</given-names></name></person-group> (<year>2016</year>). <article-title>Computer vision applied to herbarium specimens of german trees: testing the future utility of the millions of herbarium specimen images for automated identification</article-title>. <source>BMC Evol. Biol</source>. <volume>16</volume>, <fpage>248</fpage>. <pub-id pub-id-type="doi">10.1186/s12862-016-0827-5</pub-id><pub-id pub-id-type="pmid">27852219</pub-id></citation></ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>W&#x000E4;ldchen</surname> <given-names>J.</given-names></name> <name><surname>M&#x000E4;der</surname> <given-names>P.</given-names></name></person-group> (<year>2018</year>). <article-title>Plant species identification using computer vision techniques: a systematic literature review</article-title>. <source>Arch. Comput. Methods Eng</source>. <volume>25</volume>, <fpage>507</fpage>&#x02013;<lpage>543</lpage>. <pub-id pub-id-type="doi">10.1007/s11831-016-9206-z</pub-id><pub-id pub-id-type="pmid">29962832</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>F.</given-names></name> <name><surname>Cheng</surname> <given-names>J.</given-names></name> <name><surname>Liu</surname> <given-names>W.</given-names></name> <name><surname>Liu</surname> <given-names>H.</given-names></name></person-group> (<year>2018</year>). <article-title>Additive margin softmax for face verification</article-title>. <source>IEEE Signal Process. Lett</source>. <volume>25</volume>, <fpage>926</fpage>&#x02013;<lpage>930</lpage>. <pub-id pub-id-type="doi">10.1109/LSP.2018.2822810</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>Q.</given-names></name> <name><surname>Wu</surname> <given-names>B.</given-names></name> <name><surname>Zhu</surname> <given-names>P.</given-names></name> <name><surname>Li</surname> <given-names>P.</given-names></name> <name><surname>Zuo</surname> <given-names>W.</given-names></name> <name><surname>Hu</surname> <given-names>Q.</given-names></name></person-group> (<year>2020</year>). <article-title>ECA-Net: efficient channel attention for deep convolutional neural networks</article-title>, in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>. <pub-id pub-id-type="doi">10.1109/CVPR42600.2020.01155</pub-id><pub-id pub-id-type="pmid">27295638</pub-id></citation></ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wijesingha</surname> <given-names>D.</given-names></name> <name><surname>Marikar</surname> <given-names>F.</given-names></name></person-group> (<year>2012</year>). <article-title>Automatic detection system for the identification of plants using herbarium specimen images</article-title>. <source>Trop. Agric. Res</source>. <volume>23</volume>, <fpage>42</fpage>&#x02013;<lpage>50</lpage>. <pub-id pub-id-type="doi">10.4038/tar.v23i1.4630</pub-id></citation>
</ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wilf</surname> <given-names>P.</given-names></name> <name><surname>Zhang</surname> <given-names>S.</given-names></name> <name><surname>Chikkerur</surname> <given-names>S.</given-names></name> <name><surname>Little</surname> <given-names>S. A.</given-names></name> <name><surname>Wing</surname> <given-names>S. L.</given-names></name> <name><surname>Serre</surname> <given-names>T.</given-names></name></person-group> (<year>2016</year>). <article-title>Computer vision cracks the leaf code</article-title>. <source>Proc. Natl. Acad. Sci. U.S.A</source>. <volume>113</volume>, <fpage>3305</fpage>&#x02013;<lpage>3310</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1524473113</pub-id><pub-id pub-id-type="pmid">26951664</pub-id></citation></ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Willis</surname> <given-names>C. G.</given-names></name> <name><surname>Ellwood</surname> <given-names>E. R.</given-names></name> <name><surname>Primack</surname> <given-names>R. B.</given-names></name> <name><surname>Davis</surname> <given-names>C. C.</given-names></name> <name><surname>Pearson</surname> <given-names>K. D.</given-names></name> <name><surname>Gallinat</surname> <given-names>A. S.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>Old plants, new tricks: phenological research using herbarium specimens</article-title>. <source>Trends Ecol. Evol</source>. <volume>32</volume>, <fpage>531</fpage>&#x02013;<lpage>546</lpage>. <pub-id pub-id-type="doi">10.1016/j.tree.2017.03.015</pub-id><pub-id pub-id-type="pmid">28465044</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wu</surname> <given-names>S.</given-names></name> <name><surname>Manber</surname> <given-names>U.</given-names></name></person-group> (<year>1992</year>). <article-title>Fast text searching: allowing errors</article-title>. <source>Commun. ACM</source> <volume>35</volume>, <fpage>83</fpage>&#x02013;<lpage>91</lpage>. <pub-id pub-id-type="doi">10.1145/135239.135244</pub-id></citation>
</ref>
<ref id="B60">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Xie</surname> <given-names>S.</given-names></name> <name><surname>Girshick</surname> <given-names>R.</given-names></name> <name><surname>Dollar</surname> <given-names>P.</given-names></name> <name><surname>Tu</surname> <given-names>Z.</given-names></name> <name><surname>He</surname> <given-names>K.</given-names></name></person-group> (<year>2017</year>). <article-title>Aggregated residual transformations for deep neural networks</article-title>, in <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition</source>. <pub-id pub-id-type="doi">10.1109/CVPR.2017.634</pub-id><pub-id pub-id-type="pmid">31141794</pub-id></citation></ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Younis</surname> <given-names>S.</given-names></name> <name><surname>Weiland</surname> <given-names>C.</given-names></name> <name><surname>Hoehndorf</surname> <given-names>R.</given-names></name> <name><surname>Dressler</surname> <given-names>S.</given-names></name> <name><surname>Hickler</surname> <given-names>T.</given-names></name> <name><surname>Seeger</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2018</year>). <article-title>Taxon and trait recognition from digitized herbarium specimens using deep convolutional neural networks</article-title>. <source>Bot. Lett</source>. <volume>165</volume>, <fpage>377</fpage>&#x02013;<lpage>383</lpage>. <pub-id pub-id-type="doi">10.1080/23818107.2018.1446357</pub-id></citation>
</ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>H.</given-names></name> <name><surname>Wu</surname> <given-names>C.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Zhu</surname> <given-names>Y.</given-names></name> <name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Lin</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>ResNeSt: split-attention networks</article-title>. <source>arXiv preprint arXiv:2004.08955</source>.</citation>
</ref>
<ref id="B63">
<citation citation-type="web"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>X.</given-names></name> <name><surname>Yao</surname> <given-names>C.</given-names></name> <name><surname>Wen</surname> <given-names>H.</given-names></name> <name><surname>Wang</surname> <given-names>Y.</given-names></name> <name><surname>Zhou</surname> <given-names>S.</given-names></name> <name><surname>He</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>EAST: An efficient and accurate scene text detector</article-title>. <source>arXiv [Preprint].</source> <volume>arXiv</volume>: <fpage>1704.03155</fpage>. Available online at: <ext-link ext-link-type="uri" xlink:href="https://arxiv.org/pdf/1704.03155.pdf">https://arxiv.org/pdf/1704.03155.pdf</ext-link> (accessed July 10, 2017). <pub-id pub-id-type="pmid">35009864</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn id="fn0001"><p><sup>1</sup><ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com">https://www.kaggle.com</ext-link></p></fn>
<fn id="fn0002"><p><sup>2</sup><ext-link ext-link-type="uri" xlink:href="https://github.com/visipedia/herbarium_comp">https://github.com/visipedia/herbarium_comp</ext-link></p></fn>
<fn id="fn0003"><p><sup>3</sup>Images of specimens from American Samoa, Anguilla, Antigua and Barbuda, Argentina, Aruba, Australia, Bahamas, Barbados, Belize, Bermuda, Bolivia, Brazil, Canada, Cayman Islands, Chile, Colombia, Cook Islands, Costa Rica, Cuba, Cura&#x000E7;ao, Dominica, Dominican Republic, Ecuador, El Salvador, Falkland Islands, Fiji, French Guiana, French Polynesia, Greenland, Grenada, Guadeloupe, Guatemala, Guyana, Haiti, Honduras, Indonesia (island of New Guinea only), Jamaica, Kiribati, Marshall Islands, Martinique, Mexico, Micronesia, Montserrat, Nauru, New Caledonia, New Zealand, Nicaragua, Niue, Norfolk Island, Northern Mariana Islands, Palau, Panama, Papua New Guinea, Paraguay, Peru, Philippines, Pitcairn, Puerto Rico, Saint Barth&#x000E9;lemy, Saint Kitts and Nevis, Saint Lucia, Saint Martin, Saint Pierre and Miquelon, Saint Vincent and the Grenadines, Samoa, Solomon Islands, Suriname, Tokelau, Tonga, Trinidad and Tobago, Turks and Caicos Islands, Tuvalu, United States Minor Outlying Islands, United States of America, Uruguay, Vanuatu, Venezuela, Virgin Islands (both British and US), and Wallis and Futuna were included in the dataset.</p></fn>
</fn-group>
</back>
</article>
