<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//TaxonX//DTD Taxonomic Treatment Publishing DTD v0 20100105//EN" "../../nlm/tax-treatment-NS0.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:tp="http://www.plazi.org/taxpub" article-type="research-article">
  <front>
    <journal-meta>
      <journal-id journal-id-type="publisher-id">17</journal-id>
      <journal-id journal-id-type="index">urn:lsid:arphahub.com:pub:8E638694-B4E0-570A-856A-746FF325BF6B</journal-id>
      <journal-title-group>
        <journal-title xml:lang="en">Research Ideas and Outcomes</journal-title>
        <abbrev-journal-title xml:lang="en">RIO</abbrev-journal-title>
      </journal-title-group>
      <issn pub-type="epub">2367-7163</issn>
      <publisher>
        <publisher-name>Pensoft Publishers</publisher-name>
      </publisher>
    </journal-meta>
    <article-meta>
      <article-id pub-id-type="doi">10.3897/rio.8.e79187</article-id>
      <article-id pub-id-type="publisher-id">79187</article-id>
      <article-id pub-id-type="manuscript">18464</article-id>
      <article-categories>
        <subj-group subj-group-type="heading">
          <subject>Research Idea</subject>
        </subj-group>
        <subj-group subj-group-type="sdg">
          <subject>14.a Marine Biodiversity contributes to Economic Development of small/developing nations</subject>
          <subject>14.c Conservation &amp;amp; sustainable use of ocean resources</subject>
          <subject>Life below water</subject>
          <subject>Life on land</subject>
        </subj-group>
      </article-categories>
      <title-group>
        <article-title>Sharing taxonomic expertise between natural history collections using image recognition</article-title>
      </title-group>
      <contrib-group content-type="authors">
        <contrib contrib-type="author" corresp="yes">
          <name name-style="western">
            <surname>Greeff</surname>
            <given-names>Michael</given-names>
          </name>
          <email xlink:type="simple">greeffm@ethz.ch</email>
          <uri content-type="orcid">https://orcid.org/0000-0002-2697-6330</uri>
          <xref ref-type="aff" rid="A1">1</xref>
        </contrib>
        <contrib contrib-type="author" corresp="no">
          <name name-style="western">
            <surname>Caspers</surname>
            <given-names>Max</given-names>
          </name>
          <xref ref-type="aff" rid="A2">2</xref>
        </contrib>
        <contrib contrib-type="author" corresp="no">
          <name name-style="western">
            <surname>Kalkman</surname>
            <given-names>Vincent</given-names>
          </name>
          <xref ref-type="aff" rid="A2">2</xref>
        </contrib>
        <contrib contrib-type="author" corresp="no">
          <name name-style="western">
            <surname>Willemse</surname>
            <given-names>Luc</given-names>
          </name>
          <uri content-type="orcid">https://orcid.org/0000-0003-0517-9778</uri>
          <xref ref-type="aff" rid="A2">2</xref>
        </contrib>
        <contrib contrib-type="author" corresp="no">
          <name name-style="western">
            <surname>Sunderland</surname>
            <given-names>Barry Dermot</given-names>
          </name>
          <xref ref-type="aff" rid="A3">3</xref>
        </contrib>
        <contrib contrib-type="author" corresp="no">
          <name name-style="western">
            <surname>Bánki</surname>
            <given-names>Olaf</given-names>
          </name>
          <uri content-type="orcid">https://orcid.org/0000-0001-6197-9951</uri>
          <xref ref-type="aff" rid="A2">2</xref>
        </contrib>
        <contrib contrib-type="author" corresp="no">
          <name name-style="western">
            <surname>Hogeweg</surname>
            <given-names>Laurens</given-names>
          </name>
          <xref ref-type="aff" rid="A2">2</xref>
        </contrib>
      </contrib-group>
      <aff id="A1">
        <label>1</label>
        <addr-line content-type="verbatim">Department of Environmental Systems Science, ETH Zürich, Zürich, Switzerland</addr-line>
        <institution>Department of Environmental Systems Science, ETH Zürich</institution>
        <addr-line content-type="city">Zürich</addr-line>
        <country>Switzerland</country>
      </aff>
      <aff id="A2">
        <label>2</label>
        <addr-line content-type="verbatim">Naturalis Biodiversity Center, Leiden, Netherlands</addr-line>
        <institution>Naturalis Biodiversity Center</institution>
        <addr-line content-type="city">Leiden</addr-line>
        <country>Netherlands</country>
      </aff>
      <aff id="A3">
        <label>3</label>
        <addr-line content-type="verbatim">ETH Library Lab, Zürich, Switzerland</addr-line>
        <institution>ETH Library Lab</institution>
        <addr-line content-type="city">Zürich</addr-line>
        <country>Switzerland</country>
      </aff>
      <author-notes>
        <fn fn-type="corresp">
          <p>Corresponding author: Michael Greeff (<email xlink:type="simple">greeffm@ethz.ch</email>).</p>
        </fn>
        <fn fn-type="edited-by">
          <p>Academic editor: Editorial Secretary</p>
        </fn>
      </author-notes>
      <pub-date pub-type="collection">
        <year>2022</year>
      </pub-date>
      <pub-date pub-type="epub">
        <day>01</day>
        <month>03</month>
        <year>2022</year>
      </pub-date>
      <volume>8</volume>
      <elocation-id>e79187</elocation-id>
      <uri content-type="arpha" xlink:href="http://openbiodiv.net/006D1F44-FF3F-585C-9A5E-B8E7750318A1">006D1F44-FF3F-585C-9A5E-B8E7750318A1</uri>
      <history>
        <date date-type="received">
          <day>10</day>
          <month>12</month>
          <year>2021</year>
        </date>
        <date date-type="accepted">
          <day>25</day>
          <month>01</month>
          <year>2022</year>
        </date>
      </history>
      <permissions>
        <copyright-statement>Michael Greeff, Max Caspers, Vincent Kalkman, Luc Willemse, Barry Dermot Sunderland, Olaf Bánki, Laurens Hogeweg</copyright-statement>
        <license license-type="creative-commons-attribution" xlink:href="http://creativecommons.org/licenses/by/4.0/" xlink:type="simple">
          <license-p>This is an open access article distributed under the terms of the Creative Commons Attribution License (CC BY 4.0), which permits unrestricted use, distribution, and reproduction in any medium, provided the original author and source are credited.</license-p>
        </license>
      </permissions>
      <abstract>
        <label>Abstract</label>
        <p>Natural history collections play a vital role in biodiversity research and conservation by providing a window to the past. The usefulness of the vast amount of historical data depends on their quality, with correct taxonomic identifications being the most critical. The identification of many of the objects of natural history collections, however, is wanting, doubtful or outdated. Providing correct identifications is difficult given the sheer number of objects and the scarcity of expertise. Here we outline the construction of an ecosystem for the collaborative development and exchange of image recognition algorithms designed to support the identification of objects. Such an ecosystem will facilitate sharing taxonomic expertise among institutions by offering image datasets that are correctly identified by their in-house taxonomic experts. Together with openly accessible machine learning algorithms and easy to use workbenches, this will allow other institutes to train image recognition algorithms and thereby compensate for the lacking expertise.</p>
      </abstract>
      <kwd-group>
        <label>Keywords</label>
        <kwd>Digitization</kwd>
        <kwd>image recognition</kwd>
        <kwd>taxonomic expertise</kwd>
        <kwd>herbaria</kwd>
        <kwd>natural history collections</kwd>
      </kwd-group>
      <counts>
        <fig-count count="6"/>
        <table-count count="1"/>
        <ref-count count="70"/>
      </counts>
    </article-meta>
  </front>
  <body>
    <sec sec-type="Overview and background">
      <title>Overview and background</title>
      <p>Worldwide there are thousands of repositories housing natural history collections (<xref ref-type="bibr" rid="B7562911">Hobern et al. 2020</xref>) which are aggregations of preserved (parts of) biological objects. Repositories range from large national institutes with millions of specimens stored in multistory warehouses to smaller, sometimes privately owned collections. The importance of natural history collections has been highlighted from different angles in numerous papers, editorials or book chapters, for example by <xref ref-type="bibr" rid="B7563281">Suarez and Tsutsui (2004)</xref>, <xref ref-type="bibr" rid="B7562636">Bakker et al. (2020)</xref>, <xref ref-type="bibr" rid="B7563180">National Academies of Sciences, Engineering, and Medicine (2020)</xref>, or <xref ref-type="bibr" rid="B7563224">Raes et al. (2020)</xref>. The primary types housed in these collections together with published descriptions form the foundation of binomial nomenclature, enabling the anchoring of scientific names to verifiable evidence. Natural history collections also form the foundation for taxonomic classification to determine taxonomic units, such as species and higher taxon boundaries and circumscription. The collections themselves become increasingly important as windows to the past which allow us to study the impact of the Anthropocene on biodiversity (<xref ref-type="bibr" rid="B7563105">Meineke et al. 2018</xref>). The critical role natural history collections play in our society also becomes evident from the steady flow of researchers visiting repositories and the large number of research papers which use natural history collections as a primary source of data with at present nearly three peer-reviewed articles relying on data from the Global Biodiversity Information Facility GBIF being published every day (<ext-link ext-link-type="uri" xlink:href="http://www.gbif.org/literature-tracking">www.gbif.org/literature-tracking</ext-link>).</p>
      <p>
        <bold>Taxonomic identifications guarantee collection accessibility</bold>
      </p>
      <p>To make full use of natural history collections, both their physical and digital visibility and accessibility are crucial. Physical accessibility is linked to the degree of management applied to collections (for an overview of collection management levels, see <xref ref-type="bibr" rid="B7563096">McGinley (1993)</xref> or <xref ref-type="bibr" rid="B7563424">Woodburn et al. (2019)</xref>). Besides basic requirements such as climatic and sanitary conditions and ensuring minimal risks of damage, key requirements for physical accessibility are the level of identification and a transparent classification system. The Linnaean style scientific name is a key element for both physical access to specimens and online searches in biodiversity related research. Not surprisingly, the scientific name, often through the accepted taxon name, forms the central entity in most data models of biodiversity information systems to which all other information is linked. Because of the crucial role a taxon name plays in providing access to biodiversity related information, the quality of identifications that lead up to a taxon name is, or at least should be, equally important.</p>
      <p>In most repositories, collections cover large parts of the biodiversity often from all bioregions of the world. The larger the taxonomic and geographic scope of a collection the more taxonomic expertise and working time is required for its identification. For quite a while, however, there has been a trend for taxonomy to receive less and less attention in the curricula of universities, and positions in public institutions incorporating traditional taxonomy were filled with staff with no or only little taxonomic expertise. This trend, coined the taxonomic impediment (<xref ref-type="bibr" rid="B7562902">Hoagland 1996</xref>, <xref ref-type="bibr" rid="B7562926">Hopkins and Freckleton 2002</xref>), led to a diminution of taxonomic expertise available at repositories with natural history collections and a decline in their capacity to properly identify newly acquired specimens and update identification of existing specimens.</p>
      <p>Likewise, the degree of digital data capturing not only depends on capacity and funding but to a large degree also on the systematic organization of a collection, which can only be done if specimens have proper taxonomic identifications. In line with this, the Minimum Standard for Digital Specimens (MIDS), which was developed for the Distributed System of Scientific Collections DiSSCo (www.dissco.eu; <xref ref-type="bibr" rid="B7562865">Hardisty et al. 2020</xref>), considers the taxonomic identification a basic requirement with regard to digitization priorities (<xref ref-type="bibr" rid="B7562856">Hardisty 2019</xref>). Digitizing unidentified specimens seems hardly useful for most biodiversity information usages other than taxonomy itself, and digitizing wrongly identified specimens carries risk. As a result, digitization projects tend to focus on the well-sorted and thus well-known organisms, thereby neglecting large parts of collections and creating significant shortcomings and biases in our understanding of the past and present biodiversity (<xref ref-type="bibr" rid="B7563335">Troudet et al. 2017</xref>).</p>
      <p>
        <bold>Image recognition to the rescue</bold>
      </p>
      <p>As discussed in the previous sections, taxonomic knowledge is distributed very unevenly and resources for taxonomic work are scarce. For many years, there have been calls for collaboration between taxonomists and specialists in artificial intelligence, machine learning, and pattern recognition to develop automated systems capable of conducting high-throughput identification of biological specimens (<xref ref-type="bibr" rid="B7562776">Gaston and O'Neill 2004</xref>, <xref ref-type="bibr" rid="B7562950">MacLeod et al. 2010</xref>, <xref ref-type="bibr" rid="B7563382">Wäldchen et al. 2018</xref>, <xref ref-type="bibr" rid="B7562935">Høye et al. 2021</xref>). Once trained, these systems learn to distinguish objects and correctly classify them by deducing rules from a set of training data, analogous to a human brain (<xref ref-type="bibr" rid="B7563146">Mitchell 1997</xref>). Image recognition is a powerful tool to reduce the manual taxonomic workload and could be part of the solution. In this way, taxonomists will be freed from repetitive work concerning common species and put their expertise to optimal use. By using automated taxonomic identification systems, collections can upscale both their available knowledge and the range of potential staff working in collections, be it paid employees, untrained students or volunteers.</p>
      <p>Especially in the context of national and international digitization initiatives such as DiSSCo, the Integrated Digitized Biocollections iDigBio (www.idigbio.org, <xref ref-type="bibr" rid="B7562968">Matsunaga et al. 2013</xref>) or the Swiss natural history collections network SwissCollNet (<xref ref-type="bibr" rid="B7562767">Frick et al. 2019</xref>), millions of images will be created on the one hand, and taxonomic expertise will be necessary on the other hand. The ideal solution would be a machine learning solution capable of identifying all known species of the world at a high accuracy. Existing solutions such as iNaturalist (www.inaturalist.org) or Observation.org (https://waarneming.nl/apps/obsidentify) are aiming to identify all current living organisms, but they still have a strong bias for certain organismal groups and rely on quality checks by a community of human experts (<xref ref-type="bibr" rid="B7563345">Unger et al. 2020</xref>). Collection staff, however, have different needs. They want a solution focused on specimens mounted in a fixed position often already organized in taxonomic groups, for example by class or order, or originating from a geographically limited area. Compared to the current practice of sending specimens to specialists residing in institutes around the world, identification by iNaturalist would already be a much faster solution. Yet, iNaturalist still relies on a community of other users for verification which can delay the final identification by hours or days. For efficient sorting of large numbers of specimens, collection staff therefore need identifications at a very high accuracy and within seconds. In addition, collection staff often face opposite scenarios from uninformed nature lovers in the field: they do not need identifications of the commonly observed taxa, but of taxa that are less prominent in our everyday life, be it because of their lack of "beauty" or their secretive lifestyle, etc. In natural history collections, such taxa might exist in substantial numbers as collectors prefer the rare and hard to find objects. Image recognition tools trained on preferences of the average nature lover can be expected to have a strong bias for the common and to score badly on the groups found in collections (<xref ref-type="bibr" rid="B7563355">Valan 2021</xref>).</p>
      <p>Although machine learning solutions are getting ever more powerful and capable of identifying diverse objects, a single universal machine learning model for all known biological taxa is still technically challenging and costly. As a reasonable solution for the time being, collection staff therefore need machine learning tools focusing on subsets of biodiversity such as organisms from limited geographical areas and/or limited taxonomic groups. For instance, machine learning models have been developed for British ground beetle species (<xref ref-type="bibr" rid="B7562843">Hansen et al. 2019</xref>), for Palearctic butterfly species (<xref ref-type="bibr" rid="B7562738">Dhall et al. 2020</xref>, <xref ref-type="bibr" rid="B7563495">Sunderland 2020</xref>), or for closely related families of mosses (<xref ref-type="bibr" rid="B7563248">Schuettpelz et al. 2017</xref>). The authors stress that AI solutions are currently still in their initial stages and need to be treated with caution. There are many challenges such as incomplete sets of training data, geographic biases in collections or other defects. Nevertheless, as AI solutions are currently improving at a tremendous rate, the authors believe this is the right time to start integrating AI in collection management procedures and to learn from any initial obstacles.</p>
      <p>
        <bold>Automated identifications are transparent and reproducible</bold>
      </p>
      <p>Recent studies proved that machine identifications have become almost as accurate as identifications done by human experts in quite a few groups (in benthic macroinvertebrates (<xref ref-type="bibr" rid="B7562622">Ärje et al. 2020</xref>), in <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="order">Diptera</tp:taxon-name-part></tp:taxon-name>: <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="family">Chironomidae</tp:taxon-name-part></tp:taxon-name> (<xref ref-type="bibr" rid="B7563114">Milošević et al. 2020</xref>), in <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="order">Diptera</tp:taxon-name-part></tp:taxon-name> and <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="order">Coleoptera</tp:taxon-name-part></tp:taxon-name> (<xref ref-type="bibr" rid="B7563363">Valan et al. 2019</xref>), in dinoflagellates (<xref ref-type="bibr" rid="B7562728">Culverhouse et al. 2003</xref>)). Taxonomic identifications by human experts do not necessarily need to be better than machine identifications. Diverse processes and a wide range of people may be involved in the identification of each specimen in a natural history collection. Given the crucial role correct taxonomic identifications play in providing access in biodiversity related research, one should assume that a transparent evaluation system for identifications exists. Traditionally, the quality of identifications is deduced from the name of the person who performed the identification which is mentioned on an identification label. However, attaching identification labels stating details on the determiner and the taxon is time consuming and often this is omitted rendering the person and the provenance of the identification process obscure. As identifications have been carried out routinely by collection staff ranging from technicians to curators and by visiting naturalists ranging from early-stage novices to world specialists, interpreting the quality of previous identifications is far from easy. A study by <xref ref-type="bibr" rid="B7562758">Freitas et al. (2020)</xref> showed more than 23% of the <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="family">Auchenipteridae</tp:taxon-name-part></tp:taxon-name> fish records in GBIF (<ext-link ext-link-type="uri" xlink:href="http://www.gbif.org">www.gbif.org</ext-link>; <xref ref-type="bibr" rid="B7562749">Edwards 2004</xref>) and in Brazil's SpeciesLink Network (www.splink.org.br) to have inaccurate taxonomic information (this can include everything from outdated information to misidentifications). Similarly, <xref ref-type="bibr" rid="B7562822">Goodwin et al. (2015)</xref> evaluated 4,500 specimens of African gingers from 40 herbaria in 21 countries and found 58% of the specimens to have wrong names.</p>
      <p>In contrast to identifications done by human experts, machine identifications not only deliver taxonomic names, but also metadata about the probability of the determination, the range of taxa considered, the version of the application, and other parameters. Machine determinations therefore are quantifiable, transparent, and reproducible by anyone (the data management techniques involved fall under the term provenance which help reproduce, trace, assess, understand, and explain models and how they were constructed). As natural history collections data are increasingly used in statistical modeling of environmental changes and large datasets are assembled from different repositories, transparent identifications become ever more important (<xref ref-type="bibr" rid="B7563503">Souza et al. 2021</xref>).</p>
    </sec>
    <sec sec-type="Objectives">
      <title>Objectives</title>
      <p>
        <bold>An automated image recognition ecosystem</bold>
      </p>
      <p>The authors envision the establishment of a machine learning ecosystem for natural history collections which allows the sharing of existing models, image datasets and know-how between institutions and collection personnel. An avant-garde of a few experienced institutions shall develop the necessary core modules in machine learning, which can easily be re-trained by other institutions to serve their individual needs. This ecosystem should rest on four pillars:</p>
      <p><list list-type="order">
        <list-item>
          <p>a central library of machine learning algorithms and associated applications (e.g. mobile apps)</p>
        </list-item>
        <list-item>
          <p>a central library of available expert validated training datasets, be it the images themselves or simply the information where to find these images</p>
        </list-item>
        <list-item>
          <p>a digital workbench that allows even inexperienced users to customize existing machine learning solutions to their individual needs</p>
        </list-item>
        <list-item>
          <p>a user forum for the discussion of problems and the coordination of next steps, for the evaluation, testing and implementation of novel technologies, etc.</p>
        </list-item>
      </list></p>
      <p>
        <bold>Deep learning</bold>
      </p>
      <p><bold><italic>Feature extractor</italic></bold>. Deep learning models (<xref ref-type="bibr" rid="B7563290">Szegedy et al. 2015</xref>, <xref ref-type="bibr" rid="B7562832">Guo et al. 2016</xref>), the most popular in machine learning nowadays, consist of several parts, two of which are particularly important in the present context: the feature extraction network (short: feature extractor), which is sometimes also referred to as backbone, and the classifier. Well-known examples of feature extractor networks are VGG (<xref ref-type="bibr" rid="B7563614">Simonyan and Zisserman 2014</xref>), Inception (<xref ref-type="bibr" rid="B7563304">Szegedy et al. 2016</xref>) and ResNet (<xref ref-type="bibr" rid="B7562884">He et al. 2016</xref>). The feature extractor is the core of any deep learning model as it recognizes features (properties) in the signal it analyses. In case of images, the feature extractor would recognize shapes, colors, patterns etc. Feature extractors can readily be adapted to analyze other classes of objects as long as these show similar features (<xref ref-type="bibr" rid="B7563314">Tajbakhsh et al. 2016</xref>). For instance, a feature extractor trained on images of beetles can be used as well to identify images showing true bugs, cockroaches, and other morphologically similar insects. Feature extractors will therefore rarely be trained de novo, but rather be recycled in various similar contexts. The training of the feature extractor requires expert IT-knowledge, computing facilities and time and is usually done by bioinformaticians at larger institutions. An important development in recent years represents the so-called task-independent feature extractor training, in which also unlabeled images (e.g., images with unknown taxonomy) are used to extract useful features. Learning with unlabeled images is part of the field of unsupervised machine learning. A surge of recent papers (<xref ref-type="bibr" rid="B7563623">Chen et al. 2020</xref>, <xref ref-type="bibr" rid="B7563655">Caron et al. 2020</xref>) have shown that using a large number (billions) of unlabeled images can reduce the number of labeled images needed to achieve high recognition accuracy. This opens the possibility of making use of the large volumes of unlabeled material that are present in museum collections.</p>
      <p><bold><italic>Classifier</italic></bold>. The feature extractor does not relate the resulting categories to explicit human concepts such as animals, plants, or cars. For this, the machine learning model relies on a classifier network, which associates the output of the feature extractor with names and concepts (i.e., "classes"). In the natural history context, for instance, the classifier would associate certain features with a family of plants, a species of beetle etc. Classifiers can be easily (re)trained, with regard to time, computing power and experience of the user (e.g., see <xref ref-type="bibr" rid="B7563363">Valan et al. (2019)</xref> for a study in insect recognition). If for example a model existed for the <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="family">Brassicaceae</tp:taxon-name-part></tp:taxon-name> of Northern America, a simple retraining of the classifier might suffice to adapt this model to the <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="family">Brassicaceae</tp:taxon-name-part></tp:taxon-name> of Europe. This so-called transfer learning is a central deep learning concept that dramatically speeds up the training of new models and often leads to performance improvements, such as higher identification accuracy (<xref ref-type="bibr" rid="B7563674">Yosinski et al. 2014</xref>).</p>
      <p><bold><italic>Algorithms</italic></bold>. Machine learning models make predictions and are trained in a particular way and with a particular dataset as described above. Using models in practice often involves additional functionality. The complete process from image(s) to identifications can generally be described as an <italic>algorithm</italic>. Besides the models themselves, algorithms contain pre- and post-processing functionality that cannot be easily fitted into the model formalism of a feature extractor and a classifier. An example of pre-processing is explicitly localizing the organism in the picture before identification. Examples of post-processing are combining multiple predictions into one and combining image recognition models with species distribution models.</p>
      <p><bold>Central Library of Algorithms</bold>. To facilitate the exchange of these models and algorithms, the authors suggest setting up a Central Library of Algorithms (Fig. <xref ref-type="fig" rid="F7561844">1</xref>). This library would follow open data and open-source policies, provide search functionality, and allow other institutions to use the models for automatic identification after they are made available (deployed) through identification web services. For efficient utilization, the library should offer discovery as well as access and download services, a performant scalable infrastructure, an API (application programming interface) supporting machine to machine communication, and some tracking of use and accreditation services (DOIs). A comparable library or repository has been established under the name BioImage Model Zoo (www.bioimage.io), which offers community driven AI models for the analysis of mostly cellular images. Setting up a leaderboard for best performing feature extractors and algorithms will create an incentive for the generation of better training datasets and for technological improvements. For machine learning models to be useful in the daily collection work, further accompanying applications and web services are necessary, which could be shared on the central library as well. They offer user interfaces to apply the models and allow, for instance, accessing the camera and image gallery of the mobile device, cropping the images, and uploading them to the machine learning model as well as displaying the results.</p>
      <p>
        <bold>Central Library of Datasets</bold>
      </p>
      <p>Further sharing of taxonomic knowledge would be provided through a Central Library of Datasets. This library would be a system to access a collection of public image datasets for images that are suitable for supporting large-scale centralized training of feature extractors and local training of classifiers at the individual institutions (Fig. <xref ref-type="fig" rid="F7561848">2</xref>). As in the Central Library of Algorithms, this library should offer various services for users, a scalable infrastructure with interfaces and tracking functionalities. For biodiversity images, datasets may be found on GBIF and/or iDigBio. Repositories with a more general focus might be the Research Data Alliance (<xref ref-type="bibr" rid="B7563215">Parsons 2013</xref>) or the European Open Science Cloud (<ext-link ext-link-type="uri" xlink:href="http://www.eosc-portal.eu">www.eosc-portal.eu</ext-link>). In the area of machine learning, the online community of data scientists often share their datasets on Kaggle (<ext-link ext-link-type="uri" xlink:href="http://www.kaggle.com">www.kaggle.com</ext-link>). With so many collections digitizing their holdings worldwide, the number of images of specimens is growing at an impressive rate (<xref ref-type="bibr" rid="B7563326">Tegelberg et al. 2014</xref>). However, not all images are appropriate for training as many collections digitize their holdings without prior verification of the taxonomic identifications. Using these images could decrease the quality of the model but can still be used in unsupervised learning (<xref ref-type="bibr" rid="B7563623">Chen et al. 2020</xref>, <xref ref-type="bibr" rid="B7563655">Caron et al. 2020</xref>). The authors therefore suggest establishing a Central Library of Datasets to access specimen images with high confidence identifications.</p>
      <p>The uploaded images shall be collected in a dataset, in this context defined as a fixed curated list of images with additional metadata such as the name of the taxon, geographic coordinates and information on the probability of the identification. The Central Library of Datasets will reference existing public datasets such as GBIF and/or iDigBio. Over time, this can encourage collection staff and collection users to generate and publish their own datasets on public portals, possibly remedying biases and shortcomings in existing datasets (this could be done as 'data papers', see <xref ref-type="bibr" rid="B7562709">Chavan and Penev (2011)</xref>or <xref ref-type="bibr" rid="B7562718">Costello et al. (2013)</xref>). To this end, it is necessary that relevant criteria for images to qualify for training data are defined. At the BioDiversity_Next conference in Leiden in 2019 for instance, the association 'Biodiversity Information Standards TDWG' (<ext-link ext-link-type="uri" xlink:href="http://www.tdwg.org">www.tdwg.org</ext-link>) initiated the discussion on establishing a 'Deep Learning Standards Interest Group', which could take over this task. In general, the generation of interoperable training datasets will be greatly facilitated if collections start to adopt standards in all areas. The new Catalogue of Life, for example, aims to provide an authoritative nomenclature and taxonomic foundation that could function as a clearinghouse covering all scientific names of biological taxa worldwide and allows for seamless data exchange between institutions following this standard (<ext-link ext-link-type="uri" xlink:href="http://www.catalogueoflife.org">www.catalogueoflife.org</ext-link>, <xref ref-type="bibr" rid="B7562662">Bánki et al. 2018</xref>). With regard to the accuracy of taxonomic metadata, the quality of initial identifications done by human experts will be relevant. The quantification of expert knowledge as suggested by <xref ref-type="bibr" rid="B7562680">Caley et al. (2013)</xref> could prove to be a feasible solution and be used for algorithms that aggregate information and annotations (<xref ref-type="bibr" rid="B7574978">Simpson and Roberts 2015</xref>) or simply as part of one single set of criteria to define the confidence level of identifications.</p>
      <p>
        <bold>Digital workbench</bold>
      </p>
      <p>Retraining an existing model to a new group of organisms is easy – for IT specialists. The average collection manager would most likely struggle with the necessary procedures. The authors therefore propose the establishment of a digital workbench for machine learning (e.g., Google AutoML, Microsoft Azure), which allows non-experts to curate datasets (e.g., completing taxonomic or geographic information) and retrain existing models for their individual purposes. Ideally, the workbench should have a graphical user interface. Users could import existing feature extractors and further algorithms from the Central Library of Algorithms, and training data from the Central Library of Datasets (Fig. <xref ref-type="fig" rid="F7561852">3</xref>). The training of the model could be started with a few clicks, and in the end the workbench would provide a standardized evaluation of the new model informing about the accuracy and about further relevant performance indicators. When the performance is sufficient, the collection manager imports the algorithm into a mobile app or a web service, and finally may even publish it again to the Central Library of Algorithms for others to use.</p>
      <p>
        <bold>User forum</bold>
      </p>
      <p>Critical readers might consider this vision too idealistic. And it is true, for everything to work properly, many prerequisites just need to be right: a feature extractor needs to be available, appropriate images need to exist, the workbench and the applications need to work flawlessly. The authors therefore propose a further measure: the establishment of a user forum. On this forum, users can post their wishes, discuss shortcomings, and interact with more experienced institutions and providers of machine learning solutions. The user forum should thus serve as a marketplace where collection managers search for technological expertise and assistance and in return offer image datasets and taxonomic expertise. As a result, this user forum should guarantee that over time well identified image datasets and machine learning models become available for most groups of organisms, as well those that have been neglected so far. In addition, this will be the place to discuss and find strategies for shortcomings of the AI solutions related to inherent collection biases, be they geographical, cultural, taxonomical or other.</p>
    </sec>
    <sec sec-type="Use Cases">
      <title>Use Cases</title>
      <p><bold>Accessing unsorted collection holdings</bold>. Most collections accumulate considerable holdings of biological specimens which remain unidentified due to a lack of time or in-house taxonomic expertise. These specimens may be stored as singletons or as groups in boxes, either preliminarily sorted by higher taxonomic groupings (order, family) or by geographic region, or they may be completely mixed. In recent years, especially larger institutions have therefore started to database their holdings at the storage unit level (i.e., by the units in which specimens are stored, like drawers, jars, or boxes). In insect collections, for instance, whole drawers are being imaged and published online to be browsed through by the entomological community (<xref ref-type="bibr" rid="B7563206">Olsen 2015</xref>, <xref ref-type="bibr" rid="B7562959">Mantle et al. 2012</xref>). In this setting, machine learning applications could follow the image capturing step by splitting the image into segments, each of which features an individual specimen (Fig. <xref ref-type="fig" rid="F7561887">4</xref>).</p>
      <p>Machine learning applications would then recognize the taxonomic identity of each specimen (Table <xref ref-type="table" rid="T7561884">1</xref>). And finally, the collection staff can add this information to the corresponding specimen on the image, either physically using identification labels and/or digitally in the object level registration. As a result, users will be able to search for taxonomic information of individual specimens rather than of whole drawers, and collection staff will be able to efficiently sort and integrate these specimens into their main collection (Fig. <xref ref-type="fig" rid="F7561891">5</xref>).</p>
      <p><bold>Transparent identifications in mass digitization</bold>. Bringing down costs and time spent per treated item is of paramount importance when digitizing natural history collections (<xref ref-type="bibr" rid="B7561738">Blagoderov et al. 2012</xref>). In addition to introducing industrial style processes such as conveyor belts or division of labor, and recruiting volunteer workers, costs are often cut by omitting expensive work steps. In particular, natural history institutions rarely verify the taxonomic identifications of specimens prior to databasing (<xref ref-type="bibr" rid="B7563263">Scoble 2010</xref>, <xref ref-type="bibr" rid="B7563197">Oever and Gofferje 2012</xref>). If specimens are imaged, image recognition applications offer cheap and scalable means to improve data quality (Fig. <xref ref-type="fig" rid="F7561864">6</xref>). Importantly, the recognition step can be repeated at any given point in time, thus allowing not only for verification of the past identification by humans, but also for regular updates of machine identifications in the future. When a taxon is split or synonymized, for instance, the algorithm would change the taxonomic identity of the specimen in the collection management system and inform the collection manager about necessary changes in the physical collection.</p>
    </sec>
    <sec sec-type="Challenges ahead">
      <title>Challenges ahead</title>
      <p>A decade ago, the idea of using image recognition to share taxonomic knowledge between natural history collections would have seemed far-fetched. From a technical point of view this is no longer the case as is demonstrated by widely used field apps like iNaturalist, ObsIdentify (<xref ref-type="bibr" rid="B7563238">Schermer and Hogeweg 2018</xref>) or PlantNet (<xref ref-type="bibr" rid="B7562807">Goëau et al. 2011</xref>, <xref ref-type="bibr" rid="B7562794">Goëau et al. 2012</xref>). Based on expert validation, these apps have taken their place among traditional field guides and even started to replace the role of experts in identifying common species observed outdoors. The challenges for the large-scale use of image recognition in collections as described in this paper are primarily organizational and concern standardization, coordination, (re-)use of existing and development of new infrastructure components, and rallying a community of contributors and users. The premise of the outlined proposal is that collections of all sorts and sizes can have a streamlined collaboration.</p>
      <p><bold><italic>Algorithms</italic></bold>. Even though most challenges ahead are organizational, machine learning still harbors some technical challenges of its own (e.g., <xref ref-type="bibr" rid="B7562935">Høye et al. 2021</xref>). One of them is that identifications will not always be correct. Incorrect identifications with a low computed probability are relatively easy to address; they can be either discarded or the probability can be recorded along with the identification in a collection management system for future reference. Incorrect identifications with a high computed probability are a bigger problem. It can have multiple causes (<xref ref-type="bibr" rid="B7563188">Nguyen et al. 2015</xref>, <xref ref-type="bibr" rid="B7562893">Hein et al. 2019</xref>), but it occurs mainly when the dataset used to train the model differs significantly from the dataset that needs to be identified. For example, the method of preparation can be different between collections (e.g., open vs closed butterfly wings), the true taxon of a specimen being identified is not part of the original training database or there are taxa in which the distinction between species is quite difficult because of the range in biological variation. Without special measures, the output of the algorithm can be unpredictable. This so-called <italic>open world</italic> issue is well known in machine learning (<xref ref-type="bibr" rid="B7562671">Bendale and Boult 2016</xref>, <xref ref-type="bibr" rid="B7562785">Geng et al. 2021</xref>), but further study is needed to understand the difference between aleatoric (due to noise) and epistimic (due to lack of data) uncertainty in biodiversity machine learning models (<xref ref-type="bibr" rid="B7570572">Hüllermeier and Waegeman 2021</xref>).</p>
      <p><bold><italic>Standardization</italic></bold>. One organizational endeavor is to further standardize and accelerate the digitization of natural history collections, ensuring that the images and metadata can be readily applied for image recognition. This applies to both taxonomical and geographical annotations. Even when no larger infrastructure as envisioned in this paper is built, this step is worthwhile and should be addressed by or in close collaboration with TDWG (<xref ref-type="bibr" rid="B7563411">Wieczorek et al. 2012</xref>, <xref ref-type="bibr" rid="B7563154">Morris et al. 2013</xref>). It is of equal importance that the output of the models is standardized, considering aspects of accuracy and information about taxa included. It should be legible by faunistic databases, analogous to BibTeX in libraries (<ext-link ext-link-type="uri" xlink:href="http://www.bibtex.org">www.bibtex.org</ext-link>), as well as by the wide variety of collection management systems used in collections. As with manual identifications, it is necessary to record the identifier. Therefore, a commonly accepted and quotable versioning and provenance system for machine learning is necessary.</p>
      <p><bold><italic>Infrastructure</italic></bold>. Another challenge is the ownership and responsibility for the proposed ecosystem. Initially, one or several larger natural history institutions will need to build a large-scale digital infrastructure to allow for the generation, exchange, and application of image recognition models, as well as to provide a platform for a community to engage with one another. The different modules of the infrastructure can be developed by different parties. In addition, the different modules could be a combination of the repurposing of existing infrastructure components and tools and newly developed ones. Recently, a landscape and gap analysis on the automated services, tools, and workflows for extracting information from images of natural history specimens and their labels was performed (<xref ref-type="bibr" rid="B7563392">Walton et al. 2020</xref>). One could envision a similar exercise for the proposed modules. Collaborations between existing infrastructures like GBIF, Catalogue of Life, Zenodo (<ext-link ext-link-type="uri" xlink:href="https://zenodo.org">https://zenodo.org</ext-link>) amongst others, and initiatives such as DiSSCo and iDigBio could provide a framework for the repurpose of existing tooling and infrastructure components and newly developed ones. Having a standardized framework for storing and evaluating algorithms on well described datasets also provides the opportunity for the machine learning research community to compete on creating the best models in the form of challenges (<xref ref-type="bibr" rid="B7570507">Joly et al. 2020</xref>, <xref ref-type="bibr" rid="B7570530">Little et al. 2020</xref>). In the end, all modules need to fit and work together and be actively and sustainably maintained. The initial development can probably only be achieved through a grant from a national or international science foundation.</p>
      <p>Once built, the viability of the machine learning ecosystem for collections depends on the level of contribution from its participants. Collection managers and curators would need to actively focus their capacities at collaborating with experts to identify and digitize collections, resulting in taxonomically validated and properly annotated images. Once shared, they can be used to (re)train image recognition models and benefit the entire community. Especially in the initial phase this will require a level of altruism, as contributing will take time and resources while the benefits will only become clear after a few years. The concept of <italic>give and take</italic> requires momentum and should be stimulated by the collections maintaining the infrastructure, ideally utilizing already existing cross-national collaborations for mobilizing collections and knowledge. Parallels of such a community-driven approach can be found in the Barcode of Life project (<ext-link ext-link-type="uri" xlink:href="http://www.barcodinglife.org">www.barcodinglife.org</ext-link>), which allows the exchange of DNA-barcodes between institutes, or OpenML (<xref ref-type="bibr" rid="B7563373">Vanschoren et al. 2014</xref>), which facilitates the exchange and analysis of large datasets.</p>
    </sec>
  </body>
  <back>
    <ack>
      <title>Acknowledgements</title>
      <p>We are especially grateful to Rod Eastwood and Samuel Glauser for their discussion of and feedback on the current text.</p>
    </ack>
    <ref-list>
      <title>References</title>
      <ref id="B7562622">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Ärje</surname>
              <given-names>J.</given-names>
            </name>
            <name name-style="western">
              <surname>Raitoharju</surname>
              <given-names>J.</given-names>
            </name>
            <name name-style="western">
              <surname>Iosifidis</surname>
              <given-names>A.</given-names>
            </name>
            <name name-style="western">
              <surname>Tirronen</surname>
              <given-names>V.</given-names>
            </name>
            <name name-style="western">
              <surname>Meissner</surname>
              <given-names>K.</given-names>
            </name>
            <name name-style="western">
              <surname>Gabbouj</surname>
              <given-names>M.</given-names>
            </name>
            <name name-style="western">
              <surname>Kiranyaz</surname>
              <given-names>S.</given-names>
            </name>
            <name name-style="western">
              <surname>Kärkkäinen</surname>
              <given-names>S.</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Human experts vs. machines in taxa recognition</article-title>
          <source>Signal Processing: Image Communication</source>
          <volume>87</volume>
          <fpage>115917</fpage>
          <pub-id pub-id-type="doi">10.1016/j.image.2020.115917</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562636">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bakker</surname>
              <given-names>F. T.</given-names>
            </name>
            <name name-style="western">
              <surname>Antonelli</surname>
              <given-names>A.</given-names>
            </name>
            <name name-style="western">
              <surname>Clarke</surname>
              <given-names>J. A.</given-names>
            </name>
            <name name-style="western">
              <surname>Cook</surname>
              <given-names>J. A.</given-names>
            </name>
            <name name-style="western">
              <surname>Edwards</surname>
              <given-names>S. V.</given-names>
            </name>
            <name name-style="western">
              <surname>Ericson</surname>
              <given-names>P. G.P.</given-names>
            </name>
            <name name-style="western">
              <surname>Faurby</surname>
              <given-names>S.</given-names>
            </name>
            <name name-style="western">
              <surname>Ferrand</surname>
              <given-names>N.</given-names>
            </name>
            <name name-style="western">
              <surname>Gelang</surname>
              <given-names>M.</given-names>
            </name>
            <name name-style="western">
              <surname>Gillespie</surname>
              <given-names>R. G.</given-names>
            </name>
            <name name-style="western">
              <surname>Irestedt</surname>
              <given-names>M.</given-names>
            </name>
            <name name-style="western">
              <surname>Lundin</surname>
              <given-names>K.</given-names>
            </name>
            <name name-style="western">
              <surname>Larsson</surname>
              <given-names>E.</given-names>
            </name>
            <name name-style="western">
              <surname>Matos-Maraví</surname>
              <given-names>P.</given-names>
            </name>
            <name name-style="western">
              <surname>Müller</surname>
              <given-names>J.</given-names>
            </name>
            <name name-style="western">
              <surname>von Proschwitz</surname>
              <given-names>T.</given-names>
            </name>
            <name name-style="western">
              <surname>Roderick</surname>
              <given-names>G. K.</given-names>
            </name>
            <name name-style="western">
              <surname>Schliep</surname>
              <given-names>A.</given-names>
            </name>
            <name name-style="western">
              <surname>Wahlberg</surname>
              <given-names>N.</given-names>
            </name>
            <name name-style="western">
              <surname>Wiedenhoeft</surname>
              <given-names>J.</given-names>
            </name>
            <name name-style="western">
              <surname>Källersjö</surname>
              <given-names>M.</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>The Global Museum: natural history collections and the future of evolutionary science and public education.</article-title>
          <source>PeerJ</source>
          <volume>8</volume>
          <fpage>e8225</fpage>
          <pub-id pub-id-type="doi">10.7717/peerj.8225</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562662">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bánki</surname>
              <given-names>O.</given-names>
            </name>
            <name name-style="western">
              <surname>Döring</surname>
              <given-names>M.</given-names>
            </name>
            <name name-style="western">
              <surname>Holleman</surname>
              <given-names>A.</given-names>
            </name>
            <name name-style="western">
              <surname>Addink</surname>
              <given-names>W.</given-names>
            </name>
          </person-group>
          <year>2018</year>
          <article-title>Catalogue of Life Plus: innovating the CoL systems as a foundation for a clearinghouse for names and taxonomy</article-title>
          <source>Biodiversity Information Science and Standards</source>
          <volume>2</volume>
          <pub-id pub-id-type="doi">10.3897/biss.2.26922</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562671">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Bendale</surname>
              <given-names>A.</given-names>
            </name>
            <name name-style="western">
              <surname>Boult</surname>
              <given-names>T. E.</given-names>
            </name>
          </person-group>
          <year>2016</year>
          <article-title>Towards Open Set Deep Networks</article-title>
          <source>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)</source>
          <pub-id pub-id-type="doi">10.1109/cvpr.2016.173</pub-id>
        </element-citation>
      </ref>
      <ref id="B7561738">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Blagoderov</surname>
              <given-names>Vladimir</given-names>
            </name>
            <name name-style="western">
              <surname>Kitching</surname>
              <given-names>Ian</given-names>
            </name>
            <name name-style="western">
              <surname>Livermore</surname>
              <given-names>Laurence</given-names>
            </name>
            <name name-style="western">
              <surname>Simonsen</surname>
              <given-names>Thomas</given-names>
            </name>
            <name name-style="western">
              <surname>Smith</surname>
              <given-names>Vincent</given-names>
            </name>
          </person-group>
          <year>2012</year>
          <article-title>No specimen left behind: industrial scale digitization of natural history collections</article-title>
          <source>ZooKeys</source>
          <volume>209</volume>
          <fpage>133</fpage>
          <lpage>146</lpage>
          <pub-id pub-id-type="doi">10.3897/zookeys.209.3178</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562680">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Caley</surname>
              <given-names>M. J.</given-names>
            </name>
            <name name-style="western">
              <surname>O'Leary</surname>
              <given-names>R. A.</given-names>
            </name>
            <name name-style="western">
              <surname>Fisher</surname>
              <given-names>R.</given-names>
            </name>
            <name name-style="western">
              <surname>Low‐Choy</surname>
              <given-names>S.</given-names>
            </name>
            <name name-style="western">
              <surname>Johnson</surname>
              <given-names>S.</given-names>
            </name>
            <name name-style="western">
              <surname>Mengersen</surname>
              <given-names>K.</given-names>
            </name>
          </person-group>
          <year>2013</year>
          <article-title>What is an expert? A systems perspective on expertise</article-title>
          <source>Ecology and Evolution</source>
          <volume>4</volume>
          <issue>3</issue>
          <fpage>231</fpage>
          <lpage>242</lpage>
          <pub-id pub-id-type="doi">10.1002/ece3.926</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563655">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Caron</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Misra</surname>
              <given-names>I</given-names>
            </name>
            <name name-style="western">
              <surname>Mairal</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Goyal</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Bojanowski</surname>
              <given-names>P</given-names>
            </name>
            <name name-style="western">
              <surname>Joulin</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Unsupervised learning of visual features by contrasting cluster assignments</article-title>
          <source>arXiv preprint</source>
          <issue>arXiv:2006.09882</issue>
          <uri>https://arxiv.org/pdf/2006.09882.pdf</uri>
        </element-citation>
      </ref>
      <ref id="B7562700">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Caspers</surname>
              <given-names>M.</given-names>
            </name>
            <name name-style="western">
              <surname>Willemse</surname>
              <given-names>L.</given-names>
            </name>
            <name name-style="western">
              <surname>Miracle</surname>
              <given-names>E. G.</given-names>
            </name>
            <name name-style="western">
              <surname>van Nieukerken</surname>
              <given-names>E. J.</given-names>
            </name>
          </person-group>
          <year>2019</year>
          <article-title>Butterflies in bags: permanent storage of <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="order">Lepidoptera</tp:taxon-name-part></tp:taxon-name> in glassine envelopes</article-title>
          <source>Nota Lepidopterologica</source>
          <volume>42</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <lpage>16</lpage>
          <pub-id pub-id-type="doi">10.3897/nl.42.28654</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562709">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chavan</surname>
              <given-names>V.</given-names>
            </name>
            <name name-style="western">
              <surname>Penev</surname>
              <given-names>L.</given-names>
            </name>
          </person-group>
          <year>2011</year>
          <article-title>The data paper: a mechanism to incentivize data publishing in biodiversity science</article-title>
          <source>BMC Bioinformatics</source>
          <volume>12</volume>
          <pub-id pub-id-type="doi">10.1186/1471-2105-12-s15-s2</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563623">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>T</given-names>
            </name>
            <name name-style="western">
              <surname>Kornblith</surname>
              <given-names>S</given-names>
            </name>
            <name name-style="western">
              <surname>Swersky</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Norouzi</surname>
              <given-names>M</given-names>
            </name>
            <name name-style="western">
              <surname>Hinton</surname>
              <given-names>G</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Big self-supervised models are strong semi-supervised learners.</article-title>
          <source>arXiv preprint</source>
          <issue>arXiv:2006.10029</issue>
          <uri>https://arxiv.org/pdf/2006.10029.pdf</uri>
        </element-citation>
      </ref>
      <ref id="B7562718">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Costello</surname>
              <given-names>M. J.</given-names>
            </name>
            <name name-style="western">
              <surname>Michener</surname>
              <given-names>W. K.</given-names>
            </name>
            <name name-style="western">
              <surname>Gahegan</surname>
              <given-names>M.</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Z.</given-names>
            </name>
            <name name-style="western">
              <surname>Bourne</surname>
              <given-names>P. E.</given-names>
            </name>
          </person-group>
          <year>2013</year>
          <article-title>Biodiversity data should be published, cited, and peer reviewed</article-title>
          <source>Trends in Ecology &amp; Evolution</source>
          <volume>28</volume>
          <issue>8</issue>
          <fpage>454</fpage>
          <lpage>461</lpage>
          <pub-id pub-id-type="doi">10.1016/j.tree.2013.05.002</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562728">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Culverhouse</surname>
              <given-names>PF</given-names>
            </name>
            <name name-style="western">
              <surname>Williams</surname>
              <given-names>R</given-names>
            </name>
            <name name-style="western">
              <surname>Reguera</surname>
              <given-names>B</given-names>
            </name>
            <name name-style="western">
              <surname>Herry</surname>
              <given-names>V</given-names>
            </name>
            <name name-style="western">
              <surname>González-Gil</surname>
              <given-names>S</given-names>
            </name>
          </person-group>
          <year>2003</year>
          <article-title>Do experts make mistakes? A comparison of human and machine identification of dinoflagellates</article-title>
          <source>Marine Ecology Progress Series</source>
          <volume>247</volume>
          <fpage>17</fpage>
          <lpage>25</lpage>
          <pub-id pub-id-type="doi">10.3354/meps247017</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562738">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Dhall</surname>
              <given-names>Ankit</given-names>
            </name>
            <name name-style="western">
              <surname>Makarova</surname>
              <given-names>Anastasia</given-names>
            </name>
            <name name-style="western">
              <surname>Ganea</surname>
              <given-names>Octavian</given-names>
            </name>
            <name name-style="western">
              <surname>Pavllo</surname>
              <given-names>Dario</given-names>
            </name>
            <name name-style="western">
              <surname>Greeff</surname>
              <given-names>Michael</given-names>
            </name>
            <name name-style="western">
              <surname>Krause</surname>
              <given-names>Andreas</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Hierarchical Image Classification using Entailment Cone Embeddings</article-title>
          <source>2020 IEEE/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW)</source>
          <pub-id pub-id-type="doi">10.1109/cvprw50498.2020.00426</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562749">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Edwards</surname>
              <given-names>J. L.</given-names>
            </name>
          </person-group>
          <year>2004</year>
          <article-title>Research and Societal Benefits of the Global Biodiversity Information Facility</article-title>
          <source>BioScience</source>
          <volume>54</volume>
          <issue>6</issue>
          <fpage>485</fpage>
          <lpage>486</lpage>
          <pub-id pub-id-type="doi">10.1641/0006-3568(2004)054[0486:rasbot]2.0.co;2</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562758">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Freitas</surname>
              <given-names>Tiago M. S.</given-names>
            </name>
            <name name-style="western">
              <surname>Montag</surname>
              <given-names>Luciano F. A.</given-names>
            </name>
            <name name-style="western">
              <surname>De Marco</surname>
              <given-names>Paulo</given-names>
            </name>
            <name name-style="western">
              <surname>Hortal</surname>
              <given-names>JoaquÍn</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>How reliable are species identifications in biodiversity big data? Evaluating the records of a neotropical fish family in online repositories</article-title>
          <source>Systematics and Biodiversity</source>
          <volume>18</volume>
          <issue>2</issue>
          <fpage>181</fpage>
          <lpage>191</lpage>
          <pub-id pub-id-type="doi">10.1080/14772000.2020.1730473</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562767">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Frick</surname>
              <given-names>Holger</given-names>
            </name>
            <name name-style="western">
              <surname>Stieger</surname>
              <given-names>Pia</given-names>
            </name>
            <name name-style="western">
              <surname>Scheidegger</surname>
              <given-names>Christoph</given-names>
            </name>
          </person-group>
          <year>2019</year>
          <article-title>SwissCollNet – A National Initiative for Natural History Collections in Switzerland</article-title>
          <source>Biodiversity Information Science and Standards</source>
          <volume>3</volume>
          <pub-id pub-id-type="doi">10.3897/biss.3.37188</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562776">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Gaston</surname>
              <given-names>Kevin J.</given-names>
            </name>
            <name name-style="western">
              <surname>O'Neill</surname>
              <given-names>Mark A.</given-names>
            </name>
          </person-group>
          <year>2004</year>
          <article-title>Automated species identification: why not?</article-title>
          <source>Philosophical Transactions of the Royal Society of London. Series B: Biological Sciences</source>
          <volume>359</volume>
          <issue>1444</issue>
          <fpage>655</fpage>
          <lpage>667</lpage>
          <pub-id pub-id-type="doi">10.1098/rstb.2003.1442</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562785">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Geng</surname>
              <given-names>Chuanxing</given-names>
            </name>
            <name name-style="western">
              <surname>Huang</surname>
              <given-names>Sheng-Jun</given-names>
            </name>
            <name name-style="western">
              <surname>Chen</surname>
              <given-names>Songcan</given-names>
            </name>
          </person-group>
          <year>2021</year>
          <article-title>Recent Advances in Open Set Recognition: A Survey</article-title>
          <source>IEEE Transactions on Pattern Analysis and Machine Intelligence</source>
          <volume>43</volume>
          <issue>10</issue>
          <fpage>3614</fpage>
          <lpage>3631</lpage>
          <pub-id pub-id-type="doi">10.1109/tpami.2020.2981604</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562807">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Goëau</surname>
              <given-names>Hervé</given-names>
            </name>
            <name name-style="western">
              <surname>Boujemaa</surname>
              <given-names>Nozha</given-names>
            </name>
            <name name-style="western">
              <surname>Joly</surname>
              <given-names>Alexis</given-names>
            </name>
            <name name-style="western">
              <surname>Selmi</surname>
              <given-names>Souheil</given-names>
            </name>
            <name name-style="western">
              <surname>Bonnet</surname>
              <given-names>Pierre</given-names>
            </name>
            <name name-style="western">
              <surname>Mouysset</surname>
              <given-names>Elise</given-names>
            </name>
            <name name-style="western">
              <surname>Joyeux</surname>
              <given-names>Laurent</given-names>
            </name>
            <name name-style="western">
              <surname>Molino</surname>
              <given-names>Jean-François</given-names>
            </name>
            <name name-style="western">
              <surname>Birnbaum</surname>
              <given-names>Philippe</given-names>
            </name>
            <name name-style="western">
              <surname>Bathelemy</surname>
              <given-names>Daniel</given-names>
            </name>
          </person-group>
          <year>2011</year>
          <article-title>Visual-based plant species identification from crowdsourced data</article-title>
          <source>Proceedings of the 19th ACM international conference on Multimedia - MM '11</source>
          <fpage>813</fpage>
          <lpage>814</lpage>
          <pub-id pub-id-type="doi">10.1145/2072298.2072472</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562794">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Goëau</surname>
              <given-names>Hervé</given-names>
            </name>
            <name name-style="western">
              <surname>Bonnet</surname>
              <given-names>Pierre</given-names>
            </name>
            <name name-style="western">
              <surname>Barbe</surname>
              <given-names>Julien</given-names>
            </name>
            <name name-style="western">
              <surname>Bakic</surname>
              <given-names>Vera</given-names>
            </name>
            <name name-style="western">
              <surname>Joly</surname>
              <given-names>Alexis</given-names>
            </name>
            <name name-style="western">
              <surname>Molino</surname>
              <given-names>Jean-François</given-names>
            </name>
            <name name-style="western">
              <surname>Barthelemy</surname>
              <given-names>Daniel</given-names>
            </name>
            <name name-style="western">
              <surname>Boujemaa</surname>
              <given-names>Nozha</given-names>
            </name>
          </person-group>
          <year>2012</year>
          <article-title>Multi-organ plant identification</article-title>
          <source>Proceedings of the 1st ACM international workshop on Multimedia analysis for ecological data - MAED '12</source>
          <pub-id pub-id-type="doi">10.1145/2390832.2390843</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562822">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Goodwin</surname>
              <given-names>Zoë A.</given-names>
            </name>
            <name name-style="western">
              <surname>Harris</surname>
              <given-names>David J.</given-names>
            </name>
            <name name-style="western">
              <surname>Filer</surname>
              <given-names>Denis</given-names>
            </name>
            <name name-style="western">
              <surname>Wood</surname>
              <given-names>John R. I.</given-names>
            </name>
            <name name-style="western">
              <surname>Scotland</surname>
              <given-names>Robert W.</given-names>
            </name>
          </person-group>
          <year>2015</year>
          <article-title>Widespread mistaken identity in tropical plant collections</article-title>
          <source>Current Biology</source>
          <volume>25</volume>
          <issue>22</issue>
          <pub-id pub-id-type="doi">10.1016/j.cub.2015.10.002</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562832">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Guo</surname>
              <given-names>Yanming</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Yu</given-names>
            </name>
            <name name-style="western">
              <surname>Oerlemans</surname>
              <given-names>Ard</given-names>
            </name>
            <name name-style="western">
              <surname>Lao</surname>
              <given-names>Songyang</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Song</given-names>
            </name>
            <name name-style="western">
              <surname>Lew</surname>
              <given-names>Michael S.</given-names>
            </name>
          </person-group>
          <year>2016</year>
          <article-title>Deep learning for visual understanding: A review</article-title>
          <source>Neurocomputing</source>
          <volume>187</volume>
          <fpage>27</fpage>
          <lpage>48</lpage>
          <pub-id pub-id-type="doi">10.1016/j.neucom.2015.09.116</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562843">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hansen</surname>
              <given-names>Oskar L. P.</given-names>
            </name>
            <name name-style="western">
              <surname>Svenning</surname>
              <given-names>Jens‐Christian</given-names>
            </name>
            <name name-style="western">
              <surname>Olsen</surname>
              <given-names>Kent</given-names>
            </name>
            <name name-style="western">
              <surname>Dupont</surname>
              <given-names>Steen</given-names>
            </name>
            <name name-style="western">
              <surname>Garner</surname>
              <given-names>Beulah H.</given-names>
            </name>
            <name name-style="western">
              <surname>Iosifidis</surname>
              <given-names>Alexandros</given-names>
            </name>
            <name name-style="western">
              <surname>Price</surname>
              <given-names>Benjamin W.</given-names>
            </name>
            <name name-style="western">
              <surname>Høye</surname>
              <given-names>Toke T.</given-names>
            </name>
          </person-group>
          <year>2019</year>
          <article-title>Species‐level image classification with convolutional neural network enables insect identification from habitus images</article-title>
          <source>Ecology and Evolution</source>
          <volume>10</volume>
          <issue>2</issue>
          <fpage>737</fpage>
          <lpage>747</lpage>
          <pub-id pub-id-type="doi">10.1002/ece3.5921</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562856">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hardisty</surname>
              <given-names>A.</given-names>
            </name>
          </person-group>
          <year>2019</year>
          <article-title>Provisional Data Management Plan for DiSSCo infrastructure</article-title>
          <source>ICEDIG Deliverable D6.6</source>
          <pub-id pub-id-type="doi">10.5281/zenodo.3532937</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562865">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hardisty</surname>
              <given-names>Alex</given-names>
            </name>
            <name name-style="western">
              <surname>Saarenmaa</surname>
              <given-names>Hannu</given-names>
            </name>
            <name name-style="western">
              <surname>Casino</surname>
              <given-names>Ana</given-names>
            </name>
            <name name-style="western">
              <surname>Dillen</surname>
              <given-names>Mathias</given-names>
            </name>
            <name name-style="western">
              <surname>Gödderz</surname>
              <given-names>Karsten</given-names>
            </name>
            <name name-style="western">
              <surname>Groom</surname>
              <given-names>Quentin</given-names>
            </name>
            <name name-style="western">
              <surname>Hardy</surname>
              <given-names>Helen</given-names>
            </name>
            <name name-style="western">
              <surname>Koureas</surname>
              <given-names>Dimitris</given-names>
            </name>
            <name name-style="western">
              <surname>Nieva de la Hidalga</surname>
              <given-names>Abraham</given-names>
            </name>
            <name name-style="western">
              <surname>Paul</surname>
              <given-names>Deborah</given-names>
            </name>
            <name name-style="western">
              <surname>Runnel</surname>
              <given-names>Veljo</given-names>
            </name>
            <name name-style="western">
              <surname>Vermeersch</surname>
              <given-names>Xavier</given-names>
            </name>
            <name name-style="western">
              <surname>van Walsum</surname>
              <given-names>Myriam</given-names>
            </name>
            <name name-style="western">
              <surname>Willemse</surname>
              <given-names>Luc</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Conceptual design blueprint for the DiSSCo digitization infrastructure - DELIVERABLE D8.1</article-title>
          <source>Research Ideas and Outcomes</source>
          <volume>6</volume>
          <pub-id pub-id-type="doi">10.3897/rio.6.e54280</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562893">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hein</surname>
              <given-names>Matthias</given-names>
            </name>
            <name name-style="western">
              <surname>Andriushchenko</surname>
              <given-names>Maksym</given-names>
            </name>
            <name name-style="western">
              <surname>Bitterwolf</surname>
              <given-names>Julian</given-names>
            </name>
          </person-group>
          <year>2019</year>
          <article-title>Why ReLU Networks Yield High-Confidence Predictions Far Away From the Training Data and How to Mitigate the Problem</article-title>
          <source>2019 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR):</source>
          <fpage>41</fpage>
          <lpage>50</lpage>
          <pub-id pub-id-type="doi">10.1109/cvpr.2019.00013</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562884">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>He</surname>
              <given-names>Kaiming</given-names>
            </name>
            <name name-style="western">
              <surname>Zhang</surname>
              <given-names>Xiangyu</given-names>
            </name>
            <name name-style="western">
              <surname>Ren</surname>
              <given-names>Shaoqing</given-names>
            </name>
            <name name-style="western">
              <surname>Sun</surname>
              <given-names>Jian</given-names>
            </name>
          </person-group>
          <year>2016</year>
          <article-title>Deep Residual Learning for Image Recognition</article-title>
          <source>2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR):</source>
          <fpage>770</fpage>
          <lpage>778</lpage>
          <pub-id pub-id-type="doi">10.1109/cvpr.2016.90</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562902">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hoagland</surname>
              <given-names>K. E.</given-names>
            </name>
          </person-group>
          <year>1996</year>
          <article-title>The taxonomic impediment and the convention on Biodiversity</article-title>
          <source>Association of Systematics Collections Newsletter</source>
          <volume>24</volume>
          <fpage>61</fpage>
          <lpage>62</lpage>
        </element-citation>
      </ref>
      <ref id="B7562911">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hobern</surname>
              <given-names>Donald</given-names>
            </name>
            <name name-style="western">
              <surname>Paul</surname>
              <given-names>Deborah L</given-names>
            </name>
            <name name-style="western">
              <surname>Robertson</surname>
              <given-names>Tim</given-names>
            </name>
            <name name-style="western">
              <surname>Groom</surname>
              <given-names>Quentin</given-names>
            </name>
            <name name-style="western">
              <surname>Thiers</surname>
              <given-names>Barbara</given-names>
            </name>
            <name name-style="western">
              <surname>Asase</surname>
              <given-names>Alex</given-names>
            </name>
            <name name-style="western">
              <surname>Luo</surname>
              <given-names>Maofang</given-names>
            </name>
            <name name-style="western">
              <surname>Semal</surname>
              <given-names>Patrick</given-names>
            </name>
            <name name-style="western">
              <surname>Woodburn</surname>
              <given-names>Matt</given-names>
            </name>
            <name name-style="western">
              <surname>Zschuschen</surname>
              <given-names>Eliza</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Advancing the Catalogue of the World’s Natural History Collections</article-title>
          <source>Biodiversity Information Science and Standards</source>
          <volume>4</volume>
          <pub-id pub-id-type="doi">10.3897/biss.4.59324</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562926">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hopkins</surname>
              <given-names>G. W.</given-names>
            </name>
            <name name-style="western">
              <surname>Freckleton</surname>
              <given-names>R. P.</given-names>
            </name>
          </person-group>
          <year>2002</year>
          <article-title>Declines in the numbers of amateur and professional taxonomists: implications for conservation</article-title>
          <source>Animal Conservation</source>
          <volume>5</volume>
          <issue>3</issue>
          <fpage>245</fpage>
          <lpage>249</lpage>
          <pub-id pub-id-type="doi">10.1017/s1367943002002299</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562935">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Høye</surname>
              <given-names>Toke T.</given-names>
            </name>
            <name name-style="western">
              <surname>Ärje</surname>
              <given-names>Johanna</given-names>
            </name>
            <name name-style="western">
              <surname>Bjerge</surname>
              <given-names>Kim</given-names>
            </name>
            <name name-style="western">
              <surname>Hansen</surname>
              <given-names>Oskar L. P.</given-names>
            </name>
            <name name-style="western">
              <surname>Iosifidis</surname>
              <given-names>Alexandros</given-names>
            </name>
            <name name-style="western">
              <surname>Leese</surname>
              <given-names>Florian</given-names>
            </name>
            <name name-style="western">
              <surname>Mann</surname>
              <given-names>Hjalte M. R.</given-names>
            </name>
            <name name-style="western">
              <surname>Meissner</surname>
              <given-names>Kristian</given-names>
            </name>
            <name name-style="western">
              <surname>Melvad</surname>
              <given-names>Claus</given-names>
            </name>
            <name name-style="western">
              <surname>Raitoharju</surname>
              <given-names>Jenni</given-names>
            </name>
          </person-group>
          <year>2021</year>
          <article-title>Deep learning and computer vision will transform entomology</article-title>
          <source>Proceedings of the National Academy of Sciences</source>
          <volume>118</volume>
          <issue>2</issue>
          <pub-id pub-id-type="doi">10.1073/pnas.2002545117</pub-id>
        </element-citation>
      </ref>
      <ref id="B7570572">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Hüllermeier</surname>
              <given-names>Eyke</given-names>
            </name>
            <name name-style="western">
              <surname>Waegeman</surname>
              <given-names>Willem</given-names>
            </name>
          </person-group>
          <year>2021</year>
          <article-title>Aleatoric and epistemic uncertainty in machine learning: an introduction to concepts and methods</article-title>
          <source>Machine Learning</source>
          <volume>110</volume>
          <issue>3</issue>
          <fpage>457</fpage>
          <lpage>506</lpage>
          <pub-id pub-id-type="doi">10.1007/s10994-021-05946-3</pub-id>
        </element-citation>
      </ref>
      <ref id="B7570507">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Joly</surname>
              <given-names>Alexis</given-names>
            </name>
            <name name-style="western">
              <surname>Goëau</surname>
              <given-names>Hervé</given-names>
            </name>
            <name name-style="western">
              <surname>Kahl</surname>
              <given-names>Stefan</given-names>
            </name>
            <name name-style="western">
              <surname>Deneu</surname>
              <given-names>Benjamin</given-names>
            </name>
            <name name-style="western">
              <surname>Servajean</surname>
              <given-names>Maximillien</given-names>
            </name>
            <name name-style="western">
              <surname>Cole</surname>
              <given-names>Elijah</given-names>
            </name>
            <name name-style="western">
              <surname>Picek</surname>
              <given-names>Lukáš</given-names>
            </name>
            <name name-style="western">
              <surname>Ruiz de Castañeda</surname>
              <given-names>Rafael</given-names>
            </name>
            <name name-style="western">
              <surname>Bolon</surname>
              <given-names>Isabelle</given-names>
            </name>
            <name name-style="western">
              <surname>Durso</surname>
              <given-names>Andrew</given-names>
            </name>
            <name name-style="western">
              <surname>Lorieul</surname>
              <given-names>Titouan</given-names>
            </name>
            <name name-style="western">
              <surname>Botella</surname>
              <given-names>Christophe</given-names>
            </name>
            <name name-style="western">
              <surname>Glotin</surname>
              <given-names>Hervé</given-names>
            </name>
            <name name-style="western">
              <surname>Champ</surname>
              <given-names>Julien</given-names>
            </name>
            <name name-style="western">
              <surname>Eggel</surname>
              <given-names>Ivan</given-names>
            </name>
            <name name-style="western">
              <surname>Vellinga</surname>
              <given-names>Willem-Pier</given-names>
            </name>
            <name name-style="western">
              <surname>Bonnet</surname>
              <given-names>Pierre</given-names>
            </name>
            <name name-style="western">
              <surname>Müller</surname>
              <given-names>Henning</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Overview of LifeCLEF 2020: A System-Oriented Evaluation of Automated Species Identification and Species Distribution Prediction</article-title>
          <source>Lecture Notes in Computer Science</source>
          <fpage>342</fpage>
          <lpage>363</lpage>
          <pub-id pub-id-type="doi">10.1007/978-3-030-58219-7_23</pub-id>
        </element-citation>
      </ref>
      <ref id="B7570530">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Little</surname>
              <given-names>Damon P</given-names>
            </name>
            <name name-style="western">
              <surname>Tulig</surname>
              <given-names>Melissa</given-names>
            </name>
            <name name-style="western">
              <surname>Tan</surname>
              <given-names>Kiat Chuan</given-names>
            </name>
            <name name-style="western">
              <surname>Liu</surname>
              <given-names>Yulong</given-names>
            </name>
            <name name-style="western">
              <surname>Belongie</surname>
              <given-names>Serge</given-names>
            </name>
            <name name-style="western">
              <surname>Kaeser-Chen</surname>
              <given-names>Christine</given-names>
            </name>
            <name name-style="western">
              <surname>Michelangeli</surname>
              <given-names>Fabián A</given-names>
            </name>
            <name name-style="western">
              <surname>Panesar</surname>
              <given-names>Kiran</given-names>
            </name>
            <name name-style="western">
              <surname>Guha</surname>
              <given-names>R V</given-names>
            </name>
            <name name-style="western">
              <surname>Ambrose</surname>
              <given-names>Barbara A</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>An algorithm competition for automatic species identification from herbarium specimens.</article-title>
          <source>Applications in plant sciences</source>
          <volume>8</volume>
          <issue>6</issue>
          <fpage>e11365</fpage>
          <pub-id pub-id-type="doi">10.1002/aps3.11365</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562950">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>MacLeod</surname>
              <given-names>Norman</given-names>
            </name>
            <name name-style="western">
              <surname>Benfield</surname>
              <given-names>Mark</given-names>
            </name>
            <name name-style="western">
              <surname>Culverhouse</surname>
              <given-names>Phil</given-names>
            </name>
          </person-group>
          <year>2010</year>
          <article-title>Time to automate identification</article-title>
          <source>Nature</source>
          <volume>467</volume>
          <issue>7312</issue>
          <fpage>154</fpage>
          <lpage>155</lpage>
          <pub-id pub-id-type="doi">10.1038/467154a</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562959">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mantle</surname>
              <given-names>Beth</given-names>
            </name>
            <name name-style="western">
              <surname>LaSalle</surname>
              <given-names>John</given-names>
            </name>
            <name name-style="western">
              <surname>Fisher</surname>
              <given-names>Nicole</given-names>
            </name>
          </person-group>
          <year>2012</year>
          <article-title>Whole-drawer imaging for digital management and curation of a large entomological collection</article-title>
          <source>ZooKeys</source>
          <volume>209</volume>
          <fpage>147</fpage>
          <lpage>163</lpage>
          <pub-id pub-id-type="doi">10.3897/zookeys.209.3169</pub-id>
        </element-citation>
      </ref>
      <ref id="B7562968">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Matsunaga</surname>
              <given-names>Andrea</given-names>
            </name>
            <name name-style="western">
              <surname>Thompson</surname>
              <given-names>Alex</given-names>
            </name>
            <name name-style="western">
              <surname>Figueiredo</surname>
              <given-names>Renato J.</given-names>
            </name>
            <name name-style="western">
              <surname>Germain-Aubrey</surname>
              <given-names>Charlotte C.</given-names>
            </name>
            <name name-style="western">
              <surname>Collins</surname>
              <given-names>Matthew</given-names>
            </name>
            <name name-style="western">
              <surname>Beaman</surname>
              <given-names>Reed S.</given-names>
            </name>
            <name name-style="western">
              <surname>MacFadden</surname>
              <given-names>Bruce J.</given-names>
            </name>
            <name name-style="western">
              <surname>Riccardi</surname>
              <given-names>Greg</given-names>
            </name>
            <name name-style="western">
              <surname>Soltis</surname>
              <given-names>Pamela S.</given-names>
            </name>
            <name name-style="western">
              <surname>Page</surname>
              <given-names>Lawrence M.</given-names>
            </name>
            <name name-style="western">
              <surname>Fortes</surname>
              <given-names>Jose A. B.</given-names>
            </name>
          </person-group>
          <year>2013</year>
          <article-title>A Computational- and Storage-Cloud for Integration of Biodiversity Collections</article-title>
          <source>Proceedings of the IEEE 9th International Conference on e-Science:</source>
          <fpage>78</fpage>
          <lpage>87</lpage>
          <pub-id pub-id-type="doi">10.1109/escience.2013.48</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563096">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>McGinley</surname>
              <given-names>R. J.</given-names>
            </name>
          </person-group>
          <year>1993</year>
          <article-title>Where’s the management in collections management?</article-title>
          <source>International Symposium and First World Congress on the preservation and conservation of Natural History Collections</source>
          <volume>3</volume>
          <fpage>309</fpage>
          <lpage>338</lpage>
        </element-citation>
      </ref>
      <ref id="B7563105">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Meineke</surname>
              <given-names>Emily K.</given-names>
            </name>
            <name name-style="western">
              <surname>Davies</surname>
              <given-names>T. Jonathan</given-names>
            </name>
            <name name-style="western">
              <surname>Daru</surname>
              <given-names>Barnabas H.</given-names>
            </name>
            <name name-style="western">
              <surname>Davis</surname>
              <given-names>Charles C.</given-names>
            </name>
          </person-group>
          <year>2018</year>
          <article-title>Biological collections for understanding biodiversity in the Anthropocene</article-title>
          <source>Philosophical Transactions of the Royal Society B: Biological Sciences</source>
          <volume>374</volume>
          <fpage>20170386</fpage>
          <pub-id pub-id-type="doi">10.1098/rstb.2017.0386</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563114">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Milošević</surname>
              <given-names>Djuradj</given-names>
            </name>
            <name name-style="western">
              <surname>Milosavljević</surname>
              <given-names>Aleksandar</given-names>
            </name>
            <name name-style="western">
              <surname>Predić</surname>
              <given-names>Bratislav</given-names>
            </name>
            <name name-style="western">
              <surname>Medeiros</surname>
              <given-names>Andrew S.</given-names>
            </name>
            <name name-style="western">
              <surname>Savić-Zdravković</surname>
              <given-names>Dimitrija</given-names>
            </name>
            <name name-style="western">
              <surname>Stojković Piperac</surname>
              <given-names>Milica</given-names>
            </name>
            <name name-style="western">
              <surname>Kostić</surname>
              <given-names>Tijana</given-names>
            </name>
            <name name-style="western">
              <surname>Spasić</surname>
              <given-names>Filip</given-names>
            </name>
            <name name-style="western">
              <surname>Leese</surname>
              <given-names>Florian</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Application of deep learning in aquatic bioassessment: Towards automated identification of non-biting midges</article-title>
          <source>Science of The Total Environment</source>
          <volume>711</volume>
          <fpage>135160</fpage>
          <pub-id pub-id-type="doi">10.1016/j.scitotenv.2019.135160</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563146">
        <element-citation publication-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Mitchell</surname>
              <given-names>T.</given-names>
            </name>
          </person-group>
          <year>1997</year>
          <source>Machine Learning</source>
          <publisher-name>McGraw-Hill Education</publisher-name>
          <size units="page">414</size>
        </element-citation>
      </ref>
      <ref id="B7563154">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Morris</surname>
              <given-names>Robert A</given-names>
            </name>
            <name name-style="western">
              <surname>Barve</surname>
              <given-names>Vijay</given-names>
            </name>
            <name name-style="western">
              <surname>Carausu</surname>
              <given-names>Mihail</given-names>
            </name>
            <name name-style="western">
              <surname>Chavan</surname>
              <given-names>Vishwas</given-names>
            </name>
            <name name-style="western">
              <surname>Cuadra</surname>
              <given-names>José</given-names>
            </name>
            <name name-style="western">
              <surname>Freeland</surname>
              <given-names>Chris</given-names>
            </name>
            <name name-style="western">
              <surname>Hagedorn</surname>
              <given-names>Gregor</given-names>
            </name>
            <name name-style="western">
              <surname>Leary</surname>
              <given-names>Patrick</given-names>
            </name>
            <name name-style="western">
              <surname>Mozzherin</surname>
              <given-names>Dimitry</given-names>
            </name>
            <name name-style="western">
              <surname>Olson</surname>
              <given-names>Annette</given-names>
            </name>
            <name name-style="western">
              <surname>Riccardi</surname>
              <given-names>Gregory</given-names>
            </name>
            <name name-style="western">
              <surname>Teage</surname>
              <given-names>Ivan</given-names>
            </name>
            <name name-style="western">
              <surname>Whitbread</surname>
              <given-names>Greg</given-names>
            </name>
          </person-group>
          <year>2013</year>
          <article-title>Discovery and publishing of primary biodiversity data associated with multimedia resources: The Audubon Core strategies and approaches</article-title>
          <source>Biodiversity Informatics</source>
          <volume>8</volume>
          <issue>2</issue>
          <fpage>185</fpage>
          <lpage>197</lpage>
          <pub-id pub-id-type="doi">10.17161/bi.v8i2.4117</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563180">
        <element-citation publication-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>National Academies of Sciences</surname>
              <given-names>Engineering, and Medicine</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <source>Biological Collections: Ensuring Critical Research and Education for the 21st Century</source>
          <publisher-name>The National Academies Press</publisher-name>
          <publisher-loc>Washington</publisher-loc>
          <size units="page">245</size>
          <isbn>ISBN 978-0-309-49853-1</isbn>
          <pub-id pub-id-type="doi">10.17226/25592</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563188">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Nguyen</surname>
              <given-names>Anh</given-names>
            </name>
            <name name-style="western">
              <surname>Yosinski</surname>
              <given-names>Jason</given-names>
            </name>
            <name name-style="western">
              <surname>Clune</surname>
              <given-names>Jeff</given-names>
            </name>
          </person-group>
          <year>2015</year>
          <article-title>Deep neural networks are easily fooled: High confidence predictions for unrecognizable images</article-title>
          <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR):</source>
          <fpage>427</fpage>
          <lpage>436</lpage>
          <pub-id pub-id-type="doi">10.1109/cvpr.2015.7298640</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563197">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Oever</surname>
              <given-names>Jon Peter van den</given-names>
            </name>
            <name name-style="western">
              <surname>Gofferje</surname>
              <given-names>Marc</given-names>
            </name>
          </person-group>
          <year>2012</year>
          <article-title>‘From Pilot to production’: Large Scale Digitisation project at Naturalis Biodiversity Center</article-title>
          <source>ZooKeys</source>
          <volume>209</volume>
          <fpage>87</fpage>
          <lpage>92</lpage>
          <pub-id pub-id-type="doi">10.3897/zookeys.209.3609</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563206">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Olsen</surname>
              <given-names>E.</given-names>
            </name>
          </person-group>
          <year>2015</year>
          <article-title>Museum specimens find new life online</article-title>
          <source>New York Times</source>
        </element-citation>
      </ref>
      <ref id="B7563215">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Parsons</surname>
              <given-names>Mark A.</given-names>
            </name>
          </person-group>
          <year>2013</year>
          <article-title>The research data alliance: Implementing the technology, practice and connections of a data infrastructure</article-title>
          <source>Bulletin of the American Society for Information Science and Technology</source>
          <volume>39</volume>
          <issue>6</issue>
          <fpage>33</fpage>
          <lpage>36</lpage>
          <pub-id pub-id-type="doi">10.1002/bult.2013.1720390611</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563224">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Raes</surname>
              <given-names>Niels</given-names>
            </name>
            <name name-style="western">
              <surname>Casino</surname>
              <given-names>Ana</given-names>
            </name>
            <name name-style="western">
              <surname>Goodson</surname>
              <given-names>Hilary</given-names>
            </name>
            <name name-style="western">
              <surname>Islam</surname>
              <given-names>Sharif</given-names>
            </name>
            <name name-style="western">
              <surname>Koureas</surname>
              <given-names>Dimitrios</given-names>
            </name>
            <name name-style="western">
              <surname>Schiller</surname>
              <given-names>Edmund</given-names>
            </name>
            <name name-style="western">
              <surname>Schulman</surname>
              <given-names>Leif</given-names>
            </name>
            <name name-style="western">
              <surname>Tilley</surname>
              <given-names>Laura</given-names>
            </name>
            <name name-style="western">
              <surname>Robertson</surname>
              <given-names>Tim</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>White paper on the alignment and interoperability between the Distributed System of Scientific Collections (DiSSCo) and EU infrastructures - The case of the European Environment Agency (EEA)</article-title>
          <source>Research Ideas and Outcomes</source>
          <volume>6</volume>
          <pub-id pub-id-type="doi">10.3897/rio.6.e62361</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563238">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Schermer</surname>
              <given-names>Maarten</given-names>
            </name>
            <name name-style="western">
              <surname>Hogeweg</surname>
              <given-names>Laurens</given-names>
            </name>
          </person-group>
          <year>2018</year>
          <article-title>Supporting citizen scientists with automatic species identification using deep learning image recognition models</article-title>
          <source>Biodiversity Information Science and Standards</source>
          <volume>2</volume>
          <pub-id pub-id-type="doi">10.3897/biss.2.25268</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563248">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Schuettpelz</surname>
              <given-names>Eric</given-names>
            </name>
            <name name-style="western">
              <surname>Frandsen</surname>
              <given-names>Paul B</given-names>
            </name>
            <name name-style="western">
              <surname>Dikow</surname>
              <given-names>Rebecca B</given-names>
            </name>
            <name name-style="western">
              <surname>Brown</surname>
              <given-names>Abel</given-names>
            </name>
            <name name-style="western">
              <surname>Orli</surname>
              <given-names>Sylvia</given-names>
            </name>
            <name name-style="western">
              <surname>Peters</surname>
              <given-names>Melinda</given-names>
            </name>
            <name name-style="western">
              <surname>Metallo</surname>
              <given-names>Adam</given-names>
            </name>
            <name name-style="western">
              <surname>Funk</surname>
              <given-names>Vicki A</given-names>
            </name>
            <name name-style="western">
              <surname>Dorr</surname>
              <given-names>Laurence J</given-names>
            </name>
          </person-group>
          <year>2017</year>
          <article-title>Applications of deep convolutional neural networks to digitized natural history collections.</article-title>
          <source>Biodiversity data journal</source>
          <volume>5</volume>
          <fpage>e21139</fpage>
          <pub-id pub-id-type="doi">10.3897/BDJ.5.e21139</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563263">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Scoble</surname>
              <given-names>Malcolm</given-names>
            </name>
          </person-group>
          <year>2010</year>
          <article-title>Rationale and Value of Natural History Collections Digitisation</article-title>
          <source>Biodiversity Informatics</source>
          <volume>7</volume>
          <issue>2</issue>
          <fpage>77</fpage>
          <lpage>80</lpage>
          <pub-id pub-id-type="doi">10.17161/bi.v7i2.3994</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563614">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Simonyan</surname>
              <given-names>K</given-names>
            </name>
            <name name-style="western">
              <surname>Zisserman</surname>
              <given-names>A</given-names>
            </name>
          </person-group>
          <year>2014</year>
          <article-title>Very deep convolutional networks for large-scale image recognition.</article-title>
          <source>
            <italic>arXiv preprint</italic>
          </source>
          <issue>arXiv:1409.1556</issue>
          <uri>https://arxiv.org/pdf/1409.1556.pdf(2014.pdf</uri>
        </element-citation>
      </ref>
      <ref id="B7574978">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Simpson</surname>
              <given-names>Edwin</given-names>
            </name>
            <name name-style="western">
              <surname>Roberts</surname>
              <given-names>Stephen</given-names>
            </name>
          </person-group>
          <year>2015</year>
          <article-title>Bayesian Methods for Intelligent Task Assignment in Crowdsourcing Systems</article-title>
          <source>Decision Making: Uncertainty, Imperfection, Deliberation and Scalability</source>
          <fpage>1</fpage>
          <lpage>32</lpage>
          <pub-id pub-id-type="doi">10.1007/978-3-319-15144-1_1</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563503">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Souza</surname>
              <given-names>Renan</given-names>
            </name>
            <name name-style="western">
              <surname>Azevedo</surname>
              <given-names>Leonardo G.</given-names>
            </name>
            <name name-style="western">
              <surname>Lourenço</surname>
              <given-names>Vítor</given-names>
            </name>
            <name name-style="western">
              <surname>Soares</surname>
              <given-names>Elton</given-names>
            </name>
            <name name-style="western">
              <surname>Thiago</surname>
              <given-names>Raphael</given-names>
            </name>
            <name name-style="western">
              <surname>Brandão</surname>
              <given-names>Rafael</given-names>
            </name>
            <name name-style="western">
              <surname>Civitarese</surname>
              <given-names>Daniel</given-names>
            </name>
            <name name-style="western">
              <surname>Vital Brazil</surname>
              <given-names>Emilio</given-names>
            </name>
            <name name-style="western">
              <surname>Moreno</surname>
              <given-names>Marcio</given-names>
            </name>
            <name name-style="western">
              <surname>Valduriez</surname>
              <given-names>Patrick</given-names>
            </name>
            <name name-style="western">
              <surname>Mattoso</surname>
              <given-names>Marta</given-names>
            </name>
            <name name-style="western">
              <surname>Cerqueira</surname>
              <given-names>Renato</given-names>
            </name>
            <name name-style="western">
              <surname>Netto</surname>
              <given-names>Marco A. S.</given-names>
            </name>
          </person-group>
          <year>2021</year>
          <article-title>Workflow provenance in the lifecycle of scientific machine learning</article-title>
          <source>Concurrency and Computation: Practice and Experience</source>
          <pub-id pub-id-type="doi">10.1002/cpe.6544</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563281">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Suarez</surname>
              <given-names>A. V.</given-names>
            </name>
            <name name-style="western">
              <surname>Tsutsui</surname>
              <given-names>N. D.</given-names>
            </name>
          </person-group>
          <year>2004</year>
          <article-title>The Value of Museum Collections for Research and Society</article-title>
          <source>BioScience</source>
          <volume>54</volume>
          <issue>1</issue>
          <fpage>66</fpage>
          <lpage>74</lpage>
          <pub-id pub-id-type="doi">10.1641/0006-3568(2004)054[0066:tvomcf]2.0.co;2</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563495">
        <element-citation publication-type="book">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Sunderland</surname>
              <given-names>B.</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <source>BioDex: Tales from an Adventure into App Development</source>
          <publisher-name>ETH Library Lab Blog</publisher-name>
          <uri>https://www.librarylab.ethz.ch/tales-from-an-adventure-into-app-development/</uri>
        </element-citation>
      </ref>
      <ref id="B7563290">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Szegedy</surname>
              <given-names>Christian</given-names>
            </name>
            <name name-style="western">
              <surname>Wei Liu</surname>
            </name>
            <name name-style="western">
              <surname>Yangqing Jia</surname>
            </name>
            <name name-style="western">
              <surname>Sermanet</surname>
              <given-names>Pierre</given-names>
            </name>
            <name name-style="western">
              <surname>Reed</surname>
              <given-names>Scott</given-names>
            </name>
            <name name-style="western">
              <surname>Anguelov</surname>
              <given-names>Dragomir</given-names>
            </name>
            <name name-style="western">
              <surname>Erhan</surname>
              <given-names>Dumitru</given-names>
            </name>
            <name name-style="western">
              <surname>Vanhoucke</surname>
              <given-names>Vincent</given-names>
            </name>
            <name name-style="western">
              <surname>Rabinovich</surname>
              <given-names>Andrew</given-names>
            </name>
          </person-group>
          <year>2015</year>
          <article-title>Going deeper with convolutions</article-title>
          <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR):</source>
          <fpage>1</fpage>
          <lpage>9</lpage>
          <pub-id pub-id-type="doi">10.1109/cvpr.2015.7298594</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563304">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Szegedy</surname>
              <given-names>Christian</given-names>
            </name>
            <name name-style="western">
              <surname>Vanhoucke</surname>
              <given-names>Vincent</given-names>
            </name>
            <name name-style="western">
              <surname>Ioffe</surname>
              <given-names>Sergey</given-names>
            </name>
            <name name-style="western">
              <surname>Shlens</surname>
              <given-names>Jon</given-names>
            </name>
            <name name-style="western">
              <surname>Wojna</surname>
              <given-names>Zbigniew</given-names>
            </name>
          </person-group>
          <year>2016</year>
          <article-title>Rethinking the Inception Architecture for Computer Vision</article-title>
          <source>Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (CVPR):</source>
          <fpage>2818</fpage>
          <lpage>2826</lpage>
          <pub-id pub-id-type="doi">10.1109/cvpr.2016.308</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563314">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tajbakhsh</surname>
              <given-names>Nima</given-names>
            </name>
            <name name-style="western">
              <surname>Shin</surname>
              <given-names>Jae Y.</given-names>
            </name>
            <name name-style="western">
              <surname>Gurudu</surname>
              <given-names>Suryakanth R.</given-names>
            </name>
            <name name-style="western">
              <surname>Hurst</surname>
              <given-names>R. Todd</given-names>
            </name>
            <name name-style="western">
              <surname>Kendall</surname>
              <given-names>Christopher B.</given-names>
            </name>
            <name name-style="western">
              <surname>Gotway</surname>
              <given-names>Michael B.</given-names>
            </name>
            <name name-style="western">
              <surname>Liang</surname>
              <given-names>Jianming</given-names>
            </name>
          </person-group>
          <year>2016</year>
          <article-title>Convolutional Neural Networks for Medical Image Analysis: Full Training or Fine Tuning?</article-title>
          <source>IEEE Transactions on Medical Imaging</source>
          <volume>35</volume>
          <issue>5</issue>
          <fpage>1299</fpage>
          <lpage>1312</lpage>
          <pub-id pub-id-type="doi">10.1109/tmi.2016.2535302</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563326">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Tegelberg</surname>
              <given-names>Riitta</given-names>
            </name>
            <name name-style="western">
              <surname>Mononen</surname>
              <given-names>Tero</given-names>
            </name>
            <name name-style="western">
              <surname>Saarenmaa</surname>
              <given-names>Hannu</given-names>
            </name>
          </person-group>
          <year>2014</year>
          <article-title>High-performance digitization of natural history collections: Automated imaging lines for herbarium and insect specimens</article-title>
          <source>Taxon</source>
          <volume>63</volume>
          <issue>6</issue>
          <fpage>1307</fpage>
          <lpage>1313</lpage>
          <pub-id pub-id-type="doi">10.12705/636.13</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563335">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Troudet</surname>
              <given-names>Julien</given-names>
            </name>
            <name name-style="western">
              <surname>Grandcolas</surname>
              <given-names>Philippe</given-names>
            </name>
            <name name-style="western">
              <surname>Blin</surname>
              <given-names>Amandine</given-names>
            </name>
            <name name-style="western">
              <surname>Vignes-Lebbe</surname>
              <given-names>Régine</given-names>
            </name>
            <name name-style="western">
              <surname>Legendre</surname>
              <given-names>Frédéric</given-names>
            </name>
          </person-group>
          <year>2017</year>
          <article-title>Taxonomic bias in biodiversity data and societal preferences</article-title>
          <source>Scientific Reports</source>
          <volume>7</volume>
          <issue>1</issue>
          <fpage>1</fpage>
          <lpage>4</lpage>
          <pub-id pub-id-type="doi">10.1038/s41598-017-09084-6</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563345">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Unger</surname>
              <given-names>Shem</given-names>
            </name>
            <name name-style="western">
              <surname>Rollins</surname>
              <given-names>Mark</given-names>
            </name>
            <name name-style="western">
              <surname>Tietz</surname>
              <given-names>Allison</given-names>
            </name>
            <name name-style="western">
              <surname>Dumais</surname>
              <given-names>Hailey</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>iNaturalist as an engaging tool for identifying organisms in outdoor activities</article-title>
          <source>Journal of Biological Education</source>
          <fpage>1</fpage>
          <lpage>11</lpage>
          <pub-id pub-id-type="doi">10.1080/00219266.2020.1739114</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563363">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Valan</surname>
              <given-names>Miroslav</given-names>
            </name>
            <name name-style="western">
              <surname>Makonyi</surname>
              <given-names>Karoly</given-names>
            </name>
            <name name-style="western">
              <surname>Maki</surname>
              <given-names>Atsuto</given-names>
            </name>
            <name name-style="western">
              <surname>Vondráček</surname>
              <given-names>Dominik</given-names>
            </name>
            <name name-style="western">
              <surname>Ronquist</surname>
              <given-names>Fredrik</given-names>
            </name>
          </person-group>
          <year>2019</year>
          <article-title>Automated Taxonomic Identification of Insects with Expert-Level Accuracy Using Effective Feature Transfer from Convolutional Networks</article-title>
          <source>Systematic Biology</source>
          <volume>68</volume>
          <issue>6</issue>
          <fpage>876</fpage>
          <lpage>895</lpage>
          <pub-id pub-id-type="doi">10.1093/sysbio/syz014</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563355">
        <element-citation publication-type="thesis">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Valan</surname>
              <given-names>M.</given-names>
            </name>
          </person-group>
          <year>2021</year>
          <source>Automated image-based taxon identification using deep learning and citizen-science contributions. Doctoral Dissertation</source>
          <publisher-name>Department of Zoology, Stockholm University</publisher-name>
        </element-citation>
      </ref>
      <ref id="B7563373">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Vanschoren</surname>
              <given-names>Joaquin</given-names>
            </name>
            <name name-style="western">
              <surname>van Rijn</surname>
              <given-names>Jan N.</given-names>
            </name>
            <name name-style="western">
              <surname>Bischl</surname>
              <given-names>Bernd</given-names>
            </name>
            <name name-style="western">
              <surname>Torgo</surname>
              <given-names>Luis</given-names>
            </name>
          </person-group>
          <year>2014</year>
          <article-title>OpenML</article-title>
          <source>ACM SIGKDD Explorations Newsletter</source>
          <volume>15</volume>
          <issue>2</issue>
          <fpage>49</fpage>
          <lpage>60</lpage>
          <pub-id pub-id-type="doi">10.1145/2641190.2641198</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563382">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wäldchen</surname>
              <given-names>Jana</given-names>
            </name>
            <name name-style="western">
              <surname>Rzanny</surname>
              <given-names>Michael</given-names>
            </name>
            <name name-style="western">
              <surname>Seeland</surname>
              <given-names>Marco</given-names>
            </name>
            <name name-style="western">
              <surname>Mäder</surname>
              <given-names>Patrick</given-names>
            </name>
          </person-group>
          <year>2018</year>
          <article-title>Automated plant species identification—Trends and future directions</article-title>
          <source>PLOS Computational Biology</source>
          <volume>14</volume>
          <issue>4</issue>
          <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005993</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563392">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Walton</surname>
              <given-names>Stephanie</given-names>
            </name>
            <name name-style="western">
              <surname>Livermore</surname>
              <given-names>Laurence</given-names>
            </name>
            <name name-style="western">
              <surname>Bánki</surname>
              <given-names>Olaf</given-names>
            </name>
            <name name-style="western">
              <surname>Cubey</surname>
              <given-names>Robert</given-names>
            </name>
            <name name-style="western">
              <surname>Drinkwater</surname>
              <given-names>Robyn</given-names>
            </name>
            <name name-style="western">
              <surname>Englund</surname>
              <given-names>Markus</given-names>
            </name>
            <name name-style="western">
              <surname>Goble</surname>
              <given-names>Carole</given-names>
            </name>
            <name name-style="western">
              <surname>Groom</surname>
              <given-names>Quentin</given-names>
            </name>
            <name name-style="western">
              <surname>Kermorvant</surname>
              <given-names>Christopher</given-names>
            </name>
            <name name-style="western">
              <surname>Rey</surname>
              <given-names>Isabel</given-names>
            </name>
            <name name-style="western">
              <surname>Santos</surname>
              <given-names>Celia</given-names>
            </name>
            <name name-style="western">
              <surname>Scott</surname>
              <given-names>Ben</given-names>
            </name>
            <name name-style="western">
              <surname>Williams</surname>
              <given-names>Alan</given-names>
            </name>
            <name name-style="western">
              <surname>Wu</surname>
              <given-names>Zhengzhe</given-names>
            </name>
          </person-group>
          <year>2020</year>
          <article-title>Landscape Analysis for the Specimen Data Refinery</article-title>
          <source>Research Ideas and Outcomes</source>
          <volume>6</volume>
          <pub-id pub-id-type="doi">10.3897/rio.6.e57602</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563411">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Wieczorek</surname>
              <given-names>John</given-names>
            </name>
            <name name-style="western">
              <surname>Bloom</surname>
              <given-names>David</given-names>
            </name>
            <name name-style="western">
              <surname>Guralnick</surname>
              <given-names>Robert</given-names>
            </name>
            <name name-style="western">
              <surname>Blum</surname>
              <given-names>Stan</given-names>
            </name>
            <name name-style="western">
              <surname>Döring</surname>
              <given-names>Markus</given-names>
            </name>
            <name name-style="western">
              <surname>Giovanni</surname>
              <given-names>Renato</given-names>
            </name>
            <name name-style="western">
              <surname>Robertson</surname>
              <given-names>Tim</given-names>
            </name>
            <name name-style="western">
              <surname>Vieglais</surname>
              <given-names>David</given-names>
            </name>
          </person-group>
          <year>2012</year>
          <article-title>Darwin Core: An Evolving Community-Developed Biodiversity Data Standard</article-title>
          <source>PLoS ONE</source>
          <volume>7</volume>
          <issue>1</issue>
          <pub-id pub-id-type="doi">10.1371/journal.pone.0029715</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563424">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Woodburn</surname>
              <given-names>Matt</given-names>
            </name>
            <name name-style="western">
              <surname>Vincent</surname>
              <given-names>Sarah</given-names>
            </name>
            <name name-style="western">
              <surname>Hardy</surname>
              <given-names>Helen</given-names>
            </name>
            <name name-style="western">
              <surname>Valentine</surname>
              <given-names>Clare</given-names>
            </name>
          </person-group>
          <year>2019</year>
          <article-title>Join the Dots: Adding collection assessment to collection descriptions</article-title>
          <source>Biodiversity Information Science and Standards</source>
          <volume>3</volume>
          <pub-id pub-id-type="doi">10.3897/biss.3.37200</pub-id>
        </element-citation>
      </ref>
      <ref id="B7563674">
        <element-citation publication-type="article">
          <person-group person-group-type="author">
            <name name-style="western">
              <surname>Yosinski</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Clune</surname>
              <given-names>J</given-names>
            </name>
            <name name-style="western">
              <surname>Bengio</surname>
              <given-names>Y</given-names>
            </name>
            <name name-style="western">
              <surname>Lipson</surname>
              <given-names>H</given-names>
            </name>
          </person-group>
          <year>2014</year>
          <article-title>How transferable are features in deep neural networks?</article-title>
          <source>arXiv preprint</source>
          <issue>arXiv:1411.1792</issue>
          <uri>https://arxiv.org/pdf/1411.1792.pdf</uri>
        </element-citation>
      </ref>
    </ref-list>
  </back>
  <floats-group>
    <fig id="F7561844" position="float" orientation="portrait">
      <object-id content-type="arpha">AB1D9D3E-0390-59C9-95B2-7E5A41C06B10</object-id>
      <object-id content-type="doi">10.3897/rio.8.e79187.figure1</object-id>
      <label>Figure 1.</label>
      <caption>
        <p>In the Central Library of Algorithms, natural history collection staff will select algorithms (feature extractors, models, etc.) that are most appropriate for the identification of their target organisms and add them to the workbench. The current figure shows a mock-up.</p>
      </caption>
      <graphic xlink:href="rio-08-e79187-g001.jpg" position="float" id="oo_611190.jpg" orientation="portrait" xlink:type="simple">
        <uri content-type="original_file">https://binary.pensoft.net/fig/611190</uri>
      </graphic>
    </fig>
    <fig id="F7561848" position="float" orientation="portrait">
      <object-id content-type="arpha">4F2C7FAE-E475-5FDD-BA91-C9D21CA9CD1A</object-id>
      <object-id content-type="doi">10.3897/rio.8.e79187.figure2</object-id>
      <label>Figure 2.</label>
      <caption>
        <p>In the Central Library of Datasets, natural history collection staff will find correctly identified images of their target organisms and download the data for training of an individually customized classifier (photos: <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="order">Lepidoptera</tp:taxon-name-part></tp:taxon-name> by Entomological Collection of ETH Zürich; <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="order">Orthoptera</tp:taxon-name-part></tp:taxon-name> by Naturalis Biodiversity Center; <tp:taxon-name><tp:taxon-name-part taxon-name-part-type="family">Brassicaceae</tp:taxon-name-part></tp:taxon-name> by United Herbaria Z+ZT, ZT-00164967, ZT-00167494, ZT-00171530, CC BY-SA 4.0). The current figure shows a mock-up.</p>
      </caption>
      <graphic xlink:href="rio-08-e79187-g002.jpg" position="float" id="oo_611191.jpg" orientation="portrait" xlink:type="simple">
        <uri content-type="original_file">https://binary.pensoft.net/fig/611191</uri>
      </graphic>
    </fig>
    <fig id="F7561852" position="float" orientation="portrait">
      <object-id content-type="arpha">2E92D153-8C40-5BC4-92D6-5AB221015734</object-id>
      <object-id content-type="doi">10.3897/rio.8.e79187.figure3</object-id>
      <label>Figure 3.</label>
      <caption>
        <p>Sharing of taxonomic knowledge between institutes. (1) Each algorithm contains two basic components: the feature extractor and the classifier. (2) The Central Library of Datasets allows the user to browse through all available images of collection objects; (3) based on all available images, a regularly updated central feature extractor is created and published; (4) custom made algorithms can relatively easily be created by building a classifier based on a selection of taxa from the central library and combining this with the central feature extractor; (5) newly created algorithms together with their metadata (probability &amp; information on content) are published through a web service in the Central Library of Algorithms (6) and can be used through the Identification web services (API) either for batch processing of images or through a mobile app. Models can be easily extended by other institutions by combining data sources (7).</p>
      </caption>
      <graphic xlink:href="rio-08-e79187-g003.jpg" position="float" id="oo_611192.jpg" orientation="portrait" xlink:type="simple">
        <uri content-type="original_file">https://binary.pensoft.net/fig/611192</uri>
      </graphic>
    </fig>
    <fig id="F7561887" position="float" orientation="portrait">
      <object-id content-type="arpha">6187E694-E784-5346-84E9-BD9FE29C760B</object-id>
      <object-id content-type="doi">10.3897/rio.8.e79187.figure4</object-id>
      <label>Figure 4.</label>
      <caption>
        <p>Algorithms recognize and number individual specimens in a drawer of unsorted items. The insect drawer is from the Oxford University Museum of Natural History. The current figure shows a mock-up.</p>
      </caption>
      <graphic xlink:href="rio-08-e79187-g004.jpg" position="float" id="oo_611219.jpg" orientation="portrait" xlink:type="simple">
        <uri content-type="original_file">https://binary.pensoft.net/fig/611219</uri>
      </graphic>
    </fig>
    <fig id="F7561891" position="float" orientation="portrait">
      <object-id content-type="arpha">83AEA816-0E7F-5721-8DED-1744EA560168</object-id>
      <object-id content-type="doi">10.3897/rio.8.e79187.figure5</object-id>
      <label>Figure 5.</label>
      <caption>
        <p>Non-expert collection staff easily find and afterwards sort specimens by taxon (line color) and by accuracy of the identification (line type). The insect drawer is from the Oxford University Museum of Natural History. The current figure shows a mock-up.</p>
      </caption>
      <graphic xlink:href="rio-08-e79187-g005.jpg" position="float" id="oo_611220.jpg" orientation="portrait" xlink:type="simple">
        <uri content-type="original_file">https://binary.pensoft.net/fig/611220</uri>
      </graphic>
    </fig>
    <fig id="F7561864" position="float" orientation="portrait">
      <object-id content-type="arpha">C105D351-BF7B-594E-BA63-71ABAF531FBD</object-id>
      <object-id content-type="doi">10.3897/rio.8.e79187.figure6</object-id>
      <label>Figure 6.</label>
      <caption>
        <p>Mock-up of an interface for automated taxon identification. Naturalis holds over 500.000 specimens of unmounted, unsorted and often unidentified, papered butterflies and moths that were collected mostly in Europe and Asia over the past 200 years. In early 2016, Naturalis embarked on a 10-year-project to digitally identify all these specimens with the help of dedicated volunteers (<xref ref-type="bibr" rid="B7562700">Caspers et al. 2019</xref>). Specimens are unpacked, photographed, had their label data registered and then repacked, still unmounted, for long-term storage. Specimen images were then dragged and dropped into a web-based interface to get a near-instant response with multiple predictions about the taxonomic identity including probability values.</p>
      </caption>
      <graphic xlink:href="rio-08-e79187-g006.jpg" position="float" id="oo_611195.jpg" orientation="portrait" xlink:type="simple">
        <uri content-type="original_file">https://binary.pensoft.net/fig/611195</uri>
      </graphic>
    </fig>
    <table-wrap id="T7561884" position="float" orientation="portrait">
      <label>Table 1.</label>
      <caption>
        <p>Automated recognition applications identify the specimens to lower taxonomic levels and inform about the probability of the identifications.</p>
      </caption>
      <table rules="all" border="1">
        <tbody>
          <tr>
            <td rowspan="1" colspan="1">Drawer number</td>
            <td rowspan="1" colspan="1">Specimen number</td>
            <td rowspan="1" colspan="1">Family</td>
            <td rowspan="1" colspan="1">Subfamily</td>
            <td rowspan="1" colspan="1">Probability</td>
          </tr>
          <tr>
            <td rowspan="1" colspan="1">BE.2286032</td>
            <td rowspan="1" colspan="1">1</td>
            <td rowspan="1" colspan="1">
              <tp:taxon-name>
                <tp:taxon-name-part taxon-name-part-type="family">Tettigoniidae</tp:taxon-name-part>
              </tp:taxon-name>
            </td>
            <td rowspan="1" colspan="1">
              <tp:taxon-name>
                <tp:taxon-name-part taxon-name-part-type="subfamily">Conocephalinae</tp:taxon-name-part>
              </tp:taxon-name>
            </td>
            <td rowspan="1" colspan="1">95%</td>
          </tr>
          <tr>
            <td rowspan="1" colspan="1">BE.2286032</td>
            <td rowspan="1" colspan="1">2</td>
            <td rowspan="1" colspan="1">
              <tp:taxon-name>
                <tp:taxon-name-part taxon-name-part-type="family">Tettigoniidae</tp:taxon-name-part>
              </tp:taxon-name>
            </td>
            <td rowspan="1" colspan="1">
              <tp:taxon-name>
                <tp:taxon-name-part taxon-name-part-type="subfamily">Pseudophyllinae</tp:taxon-name-part>
              </tp:taxon-name>
            </td>
            <td rowspan="1" colspan="1">85%</td>
          </tr>
          <tr>
            <td rowspan="1" colspan="1">BE.2286032</td>
            <td rowspan="1" colspan="1">3</td>
            <td rowspan="1" colspan="1">
              <tp:taxon-name>
                <tp:taxon-name-part taxon-name-part-type="family">Tettigoniidae</tp:taxon-name-part>
              </tp:taxon-name>
            </td>
            <td rowspan="1" colspan="1">
              <tp:taxon-name>
                <tp:taxon-name-part taxon-name-part-type="subfamily">Pseudophyllinae</tp:taxon-name-part>
              </tp:taxon-name>
            </td>
            <td rowspan="1" colspan="1">95%</td>
          </tr>
          <tr>
            <td rowspan="1" colspan="1">...</td>
            <td rowspan="1" colspan="1">...</td>
            <td rowspan="1" colspan="1">...</td>
            <td rowspan="1" colspan="1">...</td>
            <td rowspan="1" colspan="1">...</td>
          </tr>
        </tbody>
      </table>
    </table-wrap>
  </floats-group>
</article>
